frontend.py Source File

frontend.py Source File#

Mobilint SDK qb Compiler: frontend.py Source File
Mobilint SDK qb Compiler v1.4
MCS002-KR
frontend.py
Go to the documentation of this file.
1
4
5import warnings
6from collections.abc import Callable
7from typing import Any, List, Optional, Union
8
9import torch
10from qbcompiler.compiler.execution import (
11 execute_mblt_compile_request,
12 execute_quantize_request,
13 missing_mblt_entries_message,
14)
15from qbcompiler.compiler.utils import validate_target_device
16from qbcompiler.configs import (
17 CompileConfig,
18 CalibrationConfig,
19 BitConfig,
20 ResourceManagementConfig,
21 HessianQuantConfig,
22 BiasCorrectionConfig,
23 ModConfig,
24 LlmConfig,
25 EquivalentTransformationConfig,
26 SearchWeightScaleConfig,
27 SaveSampleConfig,
28 Uint8InputConfig,
29 ExtraOutputConfig,
30 PreprocessingConfig,
31)
32from qbcompiler.model_dict.backends import (
33 MODEL_DICT_BACKENDS,
34 normalize_backend,
35)
36from qbcompiler.compile_requests import (
37 UNSET,
38 MxqCompileResolveRequest,
39 build_mblt_compile_request,
40 build_quantize_request,
41)
42from qbcompiler.artifact.input import (
43 is_existing_mblt_input,
44 is_existing_mblt_list,
45 take_parse_only_args,
46 warn_parse_args_ignored_for_mblt,
47)
48from qbcompiler.config_resolver import ConfigManager, save_compile_config
49from qbcompiler.config_resolver.deprecations import reject_removed_kwargs
50from qbcompiler.reporting.logging import get_logger
51from qbcompiler.reporting.progress import emit_progress, progress_context
52
53logger = get_logger(__name__)
54
55
56
61
62
63def mxq_compile(
64 model,
65 target_device: str,
66 calib_data_path: Union[str, List[str]] = UNSET,
67 save_subgraph_type: int = 0,
68 output_subgraph_path="",
69 save_path: Union[str, List[str]] = UNSET,
70 backend="onnx",
71 # -- ModelDict params --
72 feed_dict=None,
73 dynamic_axes=None,
74 multi_shape: Optional[dict] = None,
75 yolo_decode_include=False,
76 exclude_first_subgraph=False,
77 # -- CompileConfig-overridable (UNSET = defer to config) --
78 device=UNSET,
79 inference_scheme=UNSET,
80 use_random_calib=UNSET,
81 cpu_offload=UNSET,
82 optimize_option=UNSET,
83 buffer_mode=UNSET,
84 force_npu_input_reposition=UNSET,
85 force_npu_output_reposition=UNSET,
86 image_channels=UNSET,
87 split_blocks=UNSET,
88 split_parts=UNSET,
89 # -- Config objects --
90 config_preset: Optional[str] = None,
91 compile_config: Optional[CompileConfig] = None,
92 resource_management_config: Optional[ResourceManagementConfig] = None,
93 calibration_config: Optional[CalibrationConfig] = None,
94 bit_config: Optional[BitConfig] = None,
95 llm_config: Optional[LlmConfig] = None,
96 hessian_quant_config: Optional[HessianQuantConfig] = None,
97 mod_config: Optional[ModConfig] = None,
98 equivalent_transformation_config: Optional[EquivalentTransformationConfig] = None,
99 search_weight_scale_config: Optional[SearchWeightScaleConfig] = None,
100 save_sample_config: Optional[SaveSampleConfig] = None,
101 uint8_input_config: Optional[Uint8InputConfig] = None,
102 preprocessing_config: Optional[PreprocessingConfig] = None,
103 bias_correction=UNSET,
104 bias_correction_config: Optional[BiasCorrectionConfig] = None,
105 model_part: Optional[str] = None,
106 model_part_options: Optional[dict] = None,
107 config_save_path: Optional[str] = None,
108 *,
109 extra_output_config: Optional[ExtraOutputConfig] = None,
110 **kwargs,
111):
112 """
113 @brief Compile a model into a Mobilint eXeCUtable (MXQ) package for execution on Mobilint NPUs.
114
115 @details When no explicit value is provided for a parameter marked with @c UNSET, the default
116 from CompileConfig is used. This allows a compile_config file or object to supply the value
117 without being overridden by function-level defaults.
118
119 Configuration is resolved in priority order (highest to lowest):
120 1. Explicitly passed function arguments
121 2. Individual sub-config objects (calibration_config, llm_config, etc.)
122 3. kwargs partial overrides (quantization_method, weight_dtype, etc.)
123 4. compile_config (CompileConfig object or JSON/YAML file) or config_preset
124 5. CompileConfig field defaults
125
126 @param model string or model instance. Model path. When using @c backend="onnx", this should be the path to an ONNX
127 model file. When @c backend is "torchscript", pass the path of a @c torch.jit.save archive (it is exported to ONNX
128 first, so @p feed_dict is required); when "torch", provide a standard PyTorch model. For @c backend="tf", pass a
129 TensorFlow SavedModel directory, a Keras .keras/.h5 file or a frozen GraphDef .pb. For @c backend="tflite", pass
130 the .tflite model path. The path of an existing @c .mblt skips parsing and only quantizes.
131 A list or tuple is accepted for one purpose only: multi-shape compilation. Its entries must be existing @c .mblt
132 files that are the sizes of one model -- written by a @p multi_shape compile, or by one @c mblt_compile() call per
133 size -- and they are quantized in one call into a single multi-shape @c .mxq (then @p calib_data_path takes one
134 directory per file, in the same order). A list of source models, or of unrelated @c .mblt files, is not a valid
135 input: before quantization the files are checked to be one model at different sizes and the set is refused with the
136 first difference otherwise. Two things are compared. The source model each file records in its provenance (the
137 source file's sha256, the HuggingFace revision, ...): files made from different checkpoints are rejected even when
138 their graphs match. Only fields that pin the weights are grounds for refusal; a backend or class name that
139 disagrees is reported instead, since those also change with where the compile ran. And the graph itself: same NPU/CPU partition, same layer types in graph order (options and
140 weights are not compared, since they legitimately differ per size). A file that records no such identity -- an
141 in-memory torch model leaves only its class name -- is logged as unverified, because for it "same graph" is all
142 that was checked. To compile a source model at several sizes in one call, pass
143 one model with @p multi_shape instead.
144 @param calib_data_path string or list of strings. Path(s) to the calibration dataset. Accepts either a text/json file
145 that lists NumPy files or a directory that contains the pre-processed NumPy files.
146 @param save_subgraph_type int. Controls optional MBLT subgraph exports: 0 disables exports; 1 saves only the graph
147 structure; 2 saves graph structure plus weights; 3 saves the graph structure split into multiple subgraphs; 4 saves
148 both structure and weights split into multiple subgraphs. Defaults to 0.
149 @param output_subgraph_path string. Destination path for the exported .mblt file when @p save_subgraph_type is 1–4.
150 The resulting file can be used for visualization. Defaults to "".
151 @param save_path string or list of strings. Output MXQ filename(s). When omitted, defaults to "{model_name}.mxq"
152 derived from the model path basename.
153 @param backend string. Framework used to generate the Mobilint IR. Case-insensitive, so "ONNX" and "onnx" are
154 the same backend. "onnx", "torch", "tf" (also spelled "tensorflow" or "keras"), "tflite" and "torchscript" are all parsed by
155 the current parser. Any other value raises @c ValueError. Defaults to "onnx". When @p model is a path to an already-compiled .mblt file (the
156 output of @c mblt_compile()), it is compiled directly with MXQ regardless of the @p backend value,
157 and the file is validated beforehand; it must be a runnable
158 artifact produced by @c mblt_compile() rather than a @c save_subgraph_type preview export.
159 @param target_device string. Target NPU device for parser/compiler configuration (for example "aries-rb"). Required by execution paths that parse or quantize a model.
160 @param device string. Compilation and inference device: "cpu" or "gpu". When omitted, uses CompileConfig default.
161 @param feed_dict dict. Example input tensors for shape inference and inference validation.
162 @param multi_shape dict. Compile the model at several sizes of a dynamic dimension in one call. Keyed by input
163 name like @p feed_dict; each entry names the axis (an int, or a tuple of axes that move together) and the sizes it
164 takes: {"x": {"axis": 3, "values": [100, 200, 300]}} or {"x": {"axis": (2, 3), "values": [(224, 224), (448, 448)]}}.
165 Requires @p feed_dict: each value yields its own example tensor derived from it, and the model is parsed once per
166 value into its own .mblt. Several inputs are walked together by index, so their "values" lists must be equally long.
167 Cannot be combined with @p dynamic_axes. With @p model_part the part is built once per size from that size's
168 example tensors, so a part that sizes itself from its feed (Qwen3-ASR audio's "mel_chunk") compiles at every
169 listed size.
170 @param dynamic_axes dict. Marks model axes as dynamic. Keys are input names and values map dimension indices to
171 aliases (for example {"input": {2: "seq_len"}}).
172 @param inference_scheme string. NPU inference scheme. One of "single", "multi", "global", "global4", or
173 "global8". When omitted, uses CompileConfig default.
174 @param yolo_decode_include bool. Determines whether YOLO decode runs on NPU. Defaults to @c False.
175 @param use_random_calib bool. Generates random calibration data to validate model compilability.
176 When omitted, uses CompileConfig default.
177 @param cpu_offload bool. Enables CPU offloading during NPU inference. When omitted, uses CompileConfig default.
178 @param optimize_option int. Compiler optimization strategy selector. When omitted, uses CompileConfig default.
179 @param buffer_mode int. Buffer serialization mode: 0 uses a naive buffer, 1 uses an mmap-backed buffer.
180 When omitted, uses CompileConfig default.
181 @param force_npu_input_reposition bool. Force input reposition operations to run on NPU instead of CPU.
182 When omitted, uses CompileConfig default.
183 @param force_npu_output_reposition bool. Force output reposition operations to run on NPU instead of CPU.
184 When omitted, uses CompileConfig default.
185 @param image_channels int. Number of image channels (0 for auto-detect).
186 When omitted, uses CompileConfig default.
187 @param split_blocks list of int. Multi-MXQ split points by transformer block index.
188 Only supported for LLM models. When omitted, uses CompileConfig default.
189 @param split_parts int. Evenly split transformer blocks into N MXQ parts.
190 Only supported for LLM models. When omitted, uses CompileConfig default.
191 @param bias_correction bool. Enables in-schedule integer bias correction during weight quantization.
192 When omitted, uses @p bias_correction_config or the CompileConfig value.
193 @param exclude_first_subgraph bool. Applies only when CPU offloading is enabled: exclude the first subgraph from
194 the final graph if it is unsupported. Defaults to @c False.
195 @param config_preset string. Name of a built-in configuration preset. When provided, loads the preset
196 via CompileConfig.from_preset(). Defaults to None (no preset). Available presets:
197 - "classification": Image classification models (ResNet, EfficientNet, ViT, etc.)
198 - "detection": Object detection models (YOLO, SSD, DETR, etc.)
199 - "classification_torchvision": Torchvision classification models with standard preprocessing
200 - "yolo_640": YOLO detection models with 640x640 letterbox preprocessing
201 - "yolo_1280": YOLO detection models with 1280x1280 letterbox preprocessing
202 - "llm": Large Language Models (LLaMA, Qwen, Gemma, etc.)
203 - "llm_fast": LLM with faster compilation (less accuracy optimization)
204 - "vision_transformer": Vision Transformer models (ViT, DeiT, Swin, etc.)
205 - "multimodal": Multimodal models (CLIP, BLIP, LLaVA, etc.)
206 @param compile_config CompileConfig or string. CompileConfig object or path to JSON/YAML configuration file
207 containing all compilation settings (resourceManagement, calibration, bit, hessianQuant, mod, llm, etc.).
208 @param config_save_path string. When provided, the fully resolved configuration - the normalized
209 CompileConfig after every layer of the precedence order above has been applied, including all
210 sub-configurations - is written to this path before compilation starts. The format follows the path
211 suffix: ".yaml"/".yml" produce YAML, anything else produces JSON. Parent directories are created as
212 needed. The saved file can be fed back as @p compile_config to reproduce the same compilation.
213 Defaults to None (no config file is written).
214 @param resource_management_config ResourceManagementConfig. Resource management configuration object.
215 @param calibration_config CalibrationConfig. Calibration configuration object.
216 @param bit_config BitConfig. Bit configuration object for quantization precision settings.
217 @param llm_config LlmConfig. LLM configuration object.
218 @param hessian_quant_config HessianQuantConfig. HessianQuant (Hessian-based Quantization) configuration object.
219 @param bias_correction_config BiasCorrectionConfig. In-schedule integer bias correction settings.
220 @param model_part string. Selects a specific part of a torch model to parse ("vision", "language",
221 "encoder", ...). Typically used to support models with complex architectures (e.g. Qwen3-VL) by compiling
222 one part at a time. Parts a model declares: parser.patcher.parts.available_parts(model).
223 A model declaring exactly one part resolves it from None; one declaring several requires a name.
224 @param model_part_options dict. Extra arguments for the named part (e.g. {"mel_frames": 100} for Qwen3-ASR audio).
225 @param mod_config ModConfig. MOD (Metric-based Optimization and Distillation) configuration object.
226 @param equivalent_transformation_config EquivalentTransformationConfig. Configuration for equivalent
227 transformations like SmoothQuant.
228 @param search_weight_scale_config SearchWeightScaleConfig. Configuration for weight scale search.
229 @param save_sample_config SaveSampleConfig. Configuration for sample data generation and saving.
230 @param uint8_input_config Uint8InputConfig. Configuration for uint8 input handling.
231 @param preprocessing_config PreprocessingConfig. Preprocessing pipeline configuration.
232 @param extra_output_config ExtraOutputConfig. Promote named intermediate layers to extra
233 model outputs. Keyword-only. Names are matched against the post-fusion graph, after
234 quantization, so a layer the graph optimizer folded away is rejected by name.
235 @param kwargs dict. Additional compiler arguments. Supports partial config overrides such as
236 @c quantization_method, @c quantization_mode, @c percentile, @c weight_dtype, @c ram_usage,
237 @c max_sequence_length, etc. Also accepts deprecated parameters (@c quantization_config,
238 @c advanced_quantization_config, @c input_process_config, @c save_sample, @c sample_dtype)
239 which will emit DeprecationWarning.
240 @return None.
241
242 @par Using configuration
243 There are three ways to configure quantization settings:
244
245 1. Load all settings from a JSON/YAML config file:
246 @code{.py}
247 from qbcompiler import mxq_compile
248
249 mxq_compile(
250 model="path/to/model.onnx",
251 target_device="aries-rb",
252 calib_data_path="path/to/calib",
253 compile_config="path/to/config.json", # or config.yaml
254 device="gpu",
255 )
256 @endcode
257
258 2. Pass individual sub-config objects:
259 @code{.py}
260 from qbcompiler import mxq_compile
261 from qbcompiler.configs import (
262 ResourceManagementConfig,
263 CalibrationConfig,
264 BitConfig,
265 HessianQuantConfig,
266 ModConfig,
267 LlmConfig,
268 )
269
270 resource_mgmt = ResourceManagementConfig(weight_dtype="float32")
271 calib_cfg = CalibrationConfig(method=1, mode=1)
272 bit_cfg = BitConfig(...)
273 hessian_quant_cfg = HessianQuantConfig(apply=True)
274 mod_cfg = ModConfig(apply=False)
275 llm_cfg = LlmConfig(apply=True)
276
277 mxq_compile(
278 model="path/to/model.onnx",
279 target_device="aries-rb",
280 calib_data_path="path/to/calib",
281 resource_management_config=resource_mgmt,
282 calibration_config=calib_cfg,
283 bit_config=bit_cfg,
284 hessian_quant_config=hessian_quant_cfg,
285 mod_config=mod_cfg,
286 llm_config=llm_cfg,
287 device="gpu",
288 )
289 @endcode
290
291 3. Automatically applied partial configuration overrides:
292 @code{.py}
293 from qbcompiler import mxq_compile
294
295 mxq_compile(
296 model="path/to/model.onnx",
297 target_device="aries-rb",
298 calib_data_path="path/to/calib",
299 quantization_method=1, # per channel quantization
300 quantization_mode=1, # max percentile quantization
301 percentile=0.999, # percentile value for max percentile quantization
302 quantization_output=0, # per layer quantization for the output layer
303 device="gpu",
304 )
305 @endcode
306 Please refer to the mxq_compile function and the quantization configuration section for the meaning of the quantization-related numeric values.
307
308
309 @par Compiling models with custom inputs (ONNX/Torch/TensorFlow)
310 Provide NumPy inputs whenever the model omits shape information so that qbcompiler can infer unknown dimensions and data
311 formats.
312
313 @code{.py}
314 from qbcompiler import mxq_compile
315 import numpy as np
316
317 example_input = {
318 "input_node_name_1": np.random.randn(1, 3, 224, 224).astype(np.float32),
319 "input_node_name_2": np.random.randn(1, 8, 224, 224).astype(np.float32),
320 "input_node_name_3": np.random.randn(1, 4, 56, 56).astype(np.float32),
321 }
322
323 onnx_model_path = "path/to/your/model.onnx"
324 mxq_compile(
325 model=onnx_model_path,
326 target_device="aries-rb",
327 feed_dict=example_input,
328 backend="onnx",
329 compile_config="path/to/config.json",
330 )
331 @endcode
332 """
333 target_device = validate_target_device(target_device)
334 reject_removed_kwargs(kwargs)
335
336 if isinstance(model, (list, tuple)) or is_existing_mblt_input(model):
337 # A list is several already-parsed .mblt files: mxq_compile_from_mblt
338 # checks that every one exists, and its pipeline that they are one graph.
339 if model_part is not None or model_part_options is not None:
340 raise ValueError(
341 "model_part / model_part_options apply to parsing a live model; "
342 f"{model!r} is an already-parsed .mblt. Pass the part when "
343 "producing the mblt (mblt_compile), or pass the source model here."
344 )
345 # The parser-only arguments below are not forwarded; say so rather than
346 # letting a caller believe multi_shape or feed_dict took effect.
347 warn_parse_args_ignored_for_mblt(
348 "mxq_compile()",
349 feed_dict=feed_dict,
350 dynamic_axes=dynamic_axes,
351 multi_shape=multi_shape,
352 yolo_decode_include=yolo_decode_include,
353 exclude_first_subgraph=exclude_first_subgraph,
354 save_subgraph_type=save_subgraph_type,
355 output_subgraph_path=output_subgraph_path,
356 )
357 mxq_compile_from_mblt(
358 mblt=model,
359 calib_data_path=calib_data_path,
360 save_path=save_path,
361 backend=backend,
362 device=device,
363 inference_scheme=inference_scheme,
364 use_random_calib=use_random_calib,
365 cpu_offload=cpu_offload,
366 optimize_option=optimize_option,
367 buffer_mode=buffer_mode,
368 force_npu_input_reposition=force_npu_input_reposition,
369 force_npu_output_reposition=force_npu_output_reposition,
370 image_channels=image_channels,
371 split_blocks=split_blocks,
372 split_parts=split_parts,
373 config_preset=config_preset,
374 compile_config=compile_config,
375 resource_management_config=resource_management_config,
376 calibration_config=calibration_config,
377 bit_config=bit_config,
378 llm_config=llm_config,
379 hessian_quant_config=hessian_quant_config,
380 mod_config=mod_config,
381 equivalent_transformation_config=equivalent_transformation_config,
382 search_weight_scale_config=search_weight_scale_config,
383 save_sample_config=save_sample_config,
384 uint8_input_config=uint8_input_config,
385 preprocessing_config=preprocessing_config,
386 extra_output_config=extra_output_config,
387 bias_correction=bias_correction,
388 bias_correction_config=bias_correction_config,
389 config_save_path=config_save_path,
390 target_device=target_device,
391 **kwargs,
392 )
393 return
394
395 mxq_compile_from_source(
396 model=model,
397 calib_data_path=calib_data_path,
398 save_subgraph_type=save_subgraph_type,
399 output_subgraph_path=output_subgraph_path,
400 save_path=save_path,
401 backend=backend,
402 feed_dict=feed_dict,
403 dynamic_axes=dynamic_axes,
404 multi_shape=multi_shape,
405 yolo_decode_include=yolo_decode_include,
406 exclude_first_subgraph=exclude_first_subgraph,
407 device=device,
408 inference_scheme=inference_scheme,
409 use_random_calib=use_random_calib,
410 cpu_offload=cpu_offload,
411 optimize_option=optimize_option,
412 buffer_mode=buffer_mode,
413 force_npu_input_reposition=force_npu_input_reposition,
414 force_npu_output_reposition=force_npu_output_reposition,
415 image_channels=image_channels,
416 split_blocks=split_blocks,
417 split_parts=split_parts,
418 config_preset=config_preset,
419 compile_config=compile_config,
420 resource_management_config=resource_management_config,
421 calibration_config=calibration_config,
422 bit_config=bit_config,
423 llm_config=llm_config,
424 hessian_quant_config=hessian_quant_config,
425 mod_config=mod_config,
426 equivalent_transformation_config=equivalent_transformation_config,
427 search_weight_scale_config=search_weight_scale_config,
428 save_sample_config=save_sample_config,
429 uint8_input_config=uint8_input_config,
430 preprocessing_config=preprocessing_config,
431 extra_output_config=extra_output_config,
432 bias_correction=bias_correction,
433 bias_correction_config=bias_correction_config,
434 model_part=model_part,
435 model_part_options=model_part_options,
436 config_save_path=config_save_path,
437 target_device=target_device,
438 **kwargs,
439 )
440
441
442def mxq_compile_from_source(
443 model,
444 target_device: str,
445 calib_data_path: Union[str, List[str]] = UNSET,
446 save_subgraph_type: int = 0,
447 output_subgraph_path="",
448 save_path: Union[str, List[str]] = UNSET,
449 backend="onnx",
450 # -- ModelDict params --
451 feed_dict=None,
452 dynamic_axes=None,
453 multi_shape: Optional[dict] = None,
454 yolo_decode_include=False,
455 exclude_first_subgraph=False,
456 # -- CompileConfig-overridable (UNSET = defer to config) --
457 device=UNSET,
458 inference_scheme=UNSET,
459 use_random_calib=UNSET,
460 cpu_offload=UNSET,
461 optimize_option=UNSET,
462 buffer_mode=UNSET,
463 force_npu_input_reposition=UNSET,
464 force_npu_output_reposition=UNSET,
465 image_channels=UNSET,
466 split_blocks=UNSET,
467 split_parts=UNSET,
468 # -- Config objects --
469 config_preset: Optional[str] = None,
470 compile_config: Optional[CompileConfig] = None,
471 resource_management_config: Optional[ResourceManagementConfig] = None,
472 calibration_config: Optional[CalibrationConfig] = None,
473 bit_config: Optional[BitConfig] = None,
474 llm_config: Optional[LlmConfig] = None,
475 hessian_quant_config: Optional[HessianQuantConfig] = None,
476 mod_config: Optional[ModConfig] = None,
477 equivalent_transformation_config: Optional[EquivalentTransformationConfig] = None,
478 search_weight_scale_config: Optional[SearchWeightScaleConfig] = None,
479 save_sample_config: Optional[SaveSampleConfig] = None,
480 uint8_input_config: Optional[Uint8InputConfig] = None,
481 preprocessing_config: Optional[PreprocessingConfig] = None,
482 bias_correction=UNSET,
483 bias_correction_config: Optional[BiasCorrectionConfig] = None,
484 model_part: Optional[str] = None,
485 model_part_options: Optional[dict] = None,
486 config_save_path: Optional[str] = None,
487 *,
488 extra_output_config: Optional[ExtraOutputConfig] = None,
489 **kwargs,
490):
491 """
492 @brief Compile a raw framework model (ONNX / PyTorch / TensorFlow / TF-Lite) into an MXQ package.
493
494 @details This is the raw-model half of @c mxq_compile(): it parses the model into
495 Mobilint IR and then quantizes and compiles it in one pass. The parameters, their
496 defaults, and the configuration precedence rules are identical to @c mxq_compile() —
497 see that function for the full reference and usage examples.
498
499 Passing the path of an existing @c .mblt file here raises @c ValueError; use
500 @c mxq_compile_from_mblt() for that input instead. @c mxq_compile() routes between
501 the two automatically.
502
503 @param model string or model instance. Raw model or path to one. Must not be an
504 existing @c .mblt file.
505 @param target_device string. Target NPU device (for example "aries-rb").
506 @return None.
507 """
508 if isinstance(model, (list, tuple)) or is_existing_mblt_input(model):
509 raise ValueError(
510 "mxq_compile_from_source() does not accept an existing .mblt file. Call "
511 "mxq_compile_from_mblt() instead, or mxq_compile() to route automatically."
512 )
513
514 target_device = validate_target_device(target_device)
515 backend = normalize_backend(backend)
516 reject_removed_kwargs(kwargs)
517
518 _mxq_compile_pipeline(
519 model=model,
520 calib_data_path=calib_data_path,
521 save_subgraph_type=save_subgraph_type,
522 output_subgraph_path=output_subgraph_path,
523 save_path=save_path,
524 backend=backend,
525 target_device=target_device,
526 feed_dict=feed_dict,
527 dynamic_axes=dynamic_axes,
528 multi_shape=multi_shape,
529 yolo_decode_include=yolo_decode_include,
530 exclude_first_subgraph=exclude_first_subgraph,
531 device=device,
532 inference_scheme=inference_scheme,
533 use_random_calib=use_random_calib,
534 cpu_offload=cpu_offload,
535 optimize_option=optimize_option,
536 buffer_mode=buffer_mode,
537 force_npu_input_reposition=force_npu_input_reposition,
538 force_npu_output_reposition=force_npu_output_reposition,
539 image_channels=image_channels,
540 split_blocks=split_blocks,
541 split_parts=split_parts,
542 bias_correction=bias_correction,
543 config_preset=config_preset,
544 compile_config=compile_config,
545 config_save_path=config_save_path,
546 resource_management_config=resource_management_config,
547 calibration_config=calibration_config,
548 bit_config=bit_config,
549 llm_config=llm_config,
550 hessian_quant_config=hessian_quant_config,
551 bias_correction_config=bias_correction_config,
552 mod_config=mod_config,
553 equivalent_transformation_config=equivalent_transformation_config,
554 search_weight_scale_config=search_weight_scale_config,
555 save_sample_config=save_sample_config,
556 uint8_input_config=uint8_input_config,
557 preprocessing_config=preprocessing_config,
558 extra_output_config=extra_output_config,
559 model_part=model_part,
560 model_part_options=model_part_options,
561 **kwargs,
562 )
563
564
565def mxq_compile_from_mblt(
566 mblt: str | list[str],
567 target_device: str,
568 calib_data_path: Union[str, List[str]] = UNSET,
569 save_path: Union[str, List[str]] = UNSET,
570 backend="onnx",
571 device=UNSET,
572 inference_scheme=UNSET,
573 use_random_calib=UNSET,
574 cpu_offload=UNSET,
575 optimize_option=UNSET,
576 buffer_mode=UNSET,
577 force_npu_input_reposition=UNSET,
578 force_npu_output_reposition=UNSET,
579 image_channels=UNSET,
580 split_blocks=UNSET,
581 split_parts=UNSET,
582 config_preset: Optional[str] = None,
583 compile_config: Optional[CompileConfig] = None,
584 resource_management_config: Optional[ResourceManagementConfig] = None,
585 calibration_config: Optional[CalibrationConfig] = None,
586 bit_config: Optional[BitConfig] = None,
587 llm_config: Optional[LlmConfig] = None,
588 hessian_quant_config: Optional[HessianQuantConfig] = None,
589 mod_config: Optional[ModConfig] = None,
590 equivalent_transformation_config: Optional[EquivalentTransformationConfig] = None,
591 search_weight_scale_config: Optional[SearchWeightScaleConfig] = None,
592 save_sample_config: Optional[SaveSampleConfig] = None,
593 uint8_input_config: Optional[Uint8InputConfig] = None,
594 preprocessing_config: Optional[PreprocessingConfig] = None,
595 bias_correction=UNSET,
596 bias_correction_config: Optional[BiasCorrectionConfig] = None,
597 config_save_path: Optional[str] = None,
598 *,
599 extra_output_config: Optional[ExtraOutputConfig] = None,
600 **kwargs,
601):
602 """
603 @brief Compile an existing Mobilint IR (@c .mblt) file into an MXQ package.
604
605 @details This is the pre-parsed half of @c mxq_compile(): parsing already happened
606 (via @c mblt_compile()), so this function only quantizes and compiles. Quantization
607 parameters, their defaults, and the configuration precedence rules are identical to
608 @c mxq_compile() — see that function for the full reference and usage examples.
609
610 Parser-only arguments (@c save_subgraph_type, @c output_subgraph_path, @c feed_dict,
611 @c dynamic_axes, @c multi_shape, @c yolo_decode_include,
612 @c exclude_first_subgraph) are absent because the graph is already parsed.
613 @c mxq_compile() accepts them for backward compatibility and drops them when routing
614 here; either entry point logs a warning naming the ones that were given, so a
615 @c multi_shape or @c feed_dict that cannot apply is not mistaken for one that did.
616
617 @param mblt string or list of strings. Path to a runnable @c .mblt produced by
618 @c mblt_compile(); the file is validated up front and a @c save_subgraph_type preview
619 export is rejected. Or a list of such paths: every file is quantized in one call and
620 packed into a single @c .mxq, as a @p multi_shape compile does with the files it
621 writes per size, and @p calib_data_path takes one calibration directory per file, in
622 the same order. The files must be one graph at different sizes; before quantization
623 the pipeline checks that every file has the same NPU/CPU partition and the same layer
624 types in graph order, and refuses the set with the first difference otherwise. Options
625 and weights are not compared -- shape-dependent options and constant folding
626 legitimately differ per size.
627 @param target_device string. Target NPU device (for example "aries-rb").
628 @param backend string. Retained for backward compatibility and ignored: the graph is
629 already parsed, so no parser backend is selected.
630 @return None.
631
632 @exception ValueError @p mblt is not an existing @c .mblt file, or a list has an entry
633 that is not one; a file is not a runnable artifact for the requested @p cpu_offload
634 setting; or the files of a list are not the same graph.
635 """
636 target_device = validate_target_device(target_device)
637 reject_removed_kwargs(kwargs)
638 # This entry point has no parser-only parameters, so one passed by name
639 # lands in kwargs and would ride along into the pipeline. Take it out and
640 # report it, the same way the mxq_compile() route does.
641 warn_parse_args_ignored_for_mblt(
642 "mxq_compile_from_mblt()", **take_parse_only_args(kwargs)
643 )
644
645 if isinstance(mblt, (list, tuple)):
646 if not is_existing_mblt_list(mblt):
647 raise ValueError(missing_mblt_entries_message(mblt))
648 elif not is_existing_mblt_input(mblt):
649 raise ValueError(
650 f"mxq_compile_from_mblt() requires an existing .mblt file, got {mblt!r}. "
651 "Produce one with mblt_compile() first."
652 )
653 # The artifact itself is validated in the pipeline planner, where
654 # cpu_offload has been resolved against compile_config / config_preset.
655
656 _mxq_compile_pipeline(
657 model=mblt,
658 calib_data_path=calib_data_path,
659 save_path=save_path,
660 backend=backend,
661 device=device,
662 inference_scheme=inference_scheme,
663 use_random_calib=use_random_calib,
664 cpu_offload=cpu_offload,
665 optimize_option=optimize_option,
666 buffer_mode=buffer_mode,
667 force_npu_input_reposition=force_npu_input_reposition,
668 force_npu_output_reposition=force_npu_output_reposition,
669 image_channels=image_channels,
670 split_blocks=split_blocks,
671 split_parts=split_parts,
672 config_preset=config_preset,
673 compile_config=compile_config,
674 resource_management_config=resource_management_config,
675 calibration_config=calibration_config,
676 bit_config=bit_config,
677 llm_config=llm_config,
678 hessian_quant_config=hessian_quant_config,
679 mod_config=mod_config,
680 equivalent_transformation_config=equivalent_transformation_config,
681 search_weight_scale_config=search_weight_scale_config,
682 save_sample_config=save_sample_config,
683 uint8_input_config=uint8_input_config,
684 preprocessing_config=preprocessing_config,
685 extra_output_config=extra_output_config,
686 bias_correction=bias_correction,
687 bias_correction_config=bias_correction_config,
688 config_save_path=config_save_path,
689 target_device=target_device,
690 **kwargs,
691 )
692
693
694def mblt_compile(
695 model: str | Any,
696 mblt_save_path: str | list[str],
697 target_device: str,
698 backend="onnx",
699 device="cpu",
700 feed_dict=None,
701 dynamic_axes=None,
702 multi_shape: Optional[dict] = None,
703 yolo_decode_include=False,
704 cpu_offload=False,
705 exclude_first_subgraph=False,
706 model_part: Optional[str] = None,
707 model_part_options: Optional[dict] = None,
708 **kwargs,
709):
710 """
711 @brief Export a model to the Mobilint .mblt format without producing an MXQ package.
712
713 @param model string or model instance. Source model or path to compile.
714 @param mblt_save_path string or list of strings. Output path for the .mblt artifact; with @p multi_shape, one
715 path per value.
716 @param backend string. Framework identifier, case-insensitive. "onnx", "torch", "tf" (also spelled "tensorflow" or "keras"),
717 "tflite" and "torchscript" are all parsed by the current parser; a "torchscript" archive is exported to ONNX first
718 and needs @p feed_dict.
719 Any other value raises @c ValueError. Defaults to "onnx".
720 @param device string. Compilation device ("cpu" or "gpu"). Defaults to "cpu".
721 @param feed_dict dict. Example inputs used for shape inference.
722 @param multi_shape dict. Compile the model at several sizes of a dynamic dimension in one call. Keyed by input
723 name like @p feed_dict; each entry names the axis (an int, or a tuple of axes that move together) and the sizes it
724 takes: {"x": {"axis": 3, "values": [100, 200, 300]}} or {"x": {"axis": (2, 3), "values": [(224, 224), (448, 448)]}}.
725 Requires @p feed_dict: each value yields its own example tensor derived from it, and the model is parsed once per
726 value into its own .mblt. Several inputs are walked together by index, so their "values" lists must be equally long.
727 Pass one @p mblt_save_path per value (a list), or a single path to get <stem>_<i>.mblt.
728 Cannot be combined with @p dynamic_axes. With @p model_part the part is built once per size from that size's
729 example tensors, so a part that sizes itself from its feed (Qwen3-ASR audio's "mel_chunk") compiles at every
730 listed size.
731 @param dynamic_axes dict. Declares dynamic axes per input name.
732 @param yolo_decode_include bool. Runs YOLO decode on NPU when True.
733 @param cpu_offload bool. Enables CPU offloading for unsupported groups.
734 @param exclude_first_subgraph bool. Applies only when CPU offloading is enabled: exclude the first subgraph from the final graph if it is unsupported.
735 @param model_part string. Selects a specific part of a torch model to parse ("vision", "language",
736 "encoder", ...). Typically used to support models with complex architectures (e.g. Qwen3-VL) by compiling
737 one part at a time. Parts a model declares: parser.patcher.parts.available_parts(model).
738 A model declaring exactly one part resolves it from None; one declaring several requires a name.
739 @param model_part_options dict. Extra arguments for the named part (e.g. {"mel_frames": 100} for Qwen3-ASR audio).
740 @param kwargs dict. Additional arguments forwarded to the compiler.
741 @return None.
742 """
743 _mblt_compile_dispatch(
744 model=model,
745 mblt_save_path=mblt_save_path,
746 target_device=target_device,
747 backend=backend,
748 device=device,
749 feed_dict=feed_dict,
750 dynamic_axes=dynamic_axes,
751 multi_shape=multi_shape,
752 yolo_decode_include=yolo_decode_include,
753 cpu_offload=cpu_offload,
754 exclude_first_subgraph=exclude_first_subgraph,
755 model_part=model_part,
756 model_part_options=model_part_options,
757 **kwargs,
758 )
759
760
761def _mblt_compile_dispatch(
762 model: str | Any,
763 mblt_save_path: str | list[str],
764 target_device: str,
765 backend="onnx",
766 device=UNSET,
767 feed_dict=None,
768 dynamic_axes=None,
769 multi_shape: Optional[dict] = None,
770 yolo_decode_include=False,
771 cpu_offload=UNSET,
772 exclude_first_subgraph=False,
773 model_part: Optional[str] = None,
774 model_part_options: Optional[dict] = None,
775 **kwargs,
776) -> None:
777 """Route a compile-to-mblt call to the compile pipeline.
778
779 Shared by :func:`mblt_compile` and :func:`mblt_compile_with_callback` so
780 both validate ``target_device`` and normalize ``backend`` the same way. ``device`` / ``cpu_offload`` default to ``UNSET`` here rather
781 than to concrete values: that is what lets the callback wrapper forward only
782 the flags its caller actually set and leave the rest to the merged
783 ``CompileConfig``. :func:`mblt_compile` keeps its own historical
784 ``"cpu"`` / ``False`` defaults and passes them explicitly.
785 """
786 target_device = validate_target_device(target_device)
787 backend = normalize_backend(backend)
788 reject_removed_kwargs(kwargs)
789
790 _mblt_compile_pipeline(
791 model=model,
792 mblt_save_path=mblt_save_path,
793 backend=backend,
794 target_device=target_device,
795 device=device,
796 feed_dict=feed_dict,
797 dynamic_axes=dynamic_axes,
798 multi_shape=multi_shape,
799 yolo_decode_include=yolo_decode_include,
800 cpu_offload=cpu_offload,
801 exclude_first_subgraph=exclude_first_subgraph,
802 model_part=model_part,
803 model_part_options=model_part_options,
804 **kwargs,
805 )
806
807
808def _mxq_compile_pipeline(
809 model,
810 target_device: str,
811 calib_data_path: Union[str, List[str]] = UNSET,
812 save_subgraph_type: int = 0,
813 output_subgraph_path="",
814 save_path: Union[str, List[str]] = UNSET,
815 backend="onnx",
816 feed_dict=None,
817 dynamic_axes=None,
818 multi_shape: Optional[dict] = None,
819 yolo_decode_include=False,
820 exclude_first_subgraph=False,
821 device=UNSET,
822 inference_scheme=UNSET,
823 use_random_calib=UNSET,
824 cpu_offload=UNSET,
825 optimize_option=UNSET,
826 buffer_mode=UNSET,
827 force_npu_input_reposition=UNSET,
828 force_npu_output_reposition=UNSET,
829 image_channels=UNSET,
830 split_blocks=UNSET,
831 split_parts=UNSET,
832 config_preset: Optional[str] = None,
833 compile_config: Optional[CompileConfig | str] = None,
834 resource_management_config: Optional[ResourceManagementConfig] = None,
835 calibration_config: Optional[CalibrationConfig] = None,
836 bit_config: Optional[BitConfig] = None,
837 llm_config: Optional[LlmConfig] = None,
838 hessian_quant_config: Optional[HessianQuantConfig] = None,
839 mod_config: Optional[ModConfig] = None,
840 equivalent_transformation_config: Optional[EquivalentTransformationConfig] = None,
841 search_weight_scale_config: Optional[SearchWeightScaleConfig] = None,
842 save_sample_config: Optional[SaveSampleConfig] = None,
843 uint8_input_config: Optional[Uint8InputConfig] = None,
844 preprocessing_config: Optional[PreprocessingConfig] = None,
845 bias_correction=UNSET,
846 bias_correction_config: Optional[BiasCorrectionConfig] = None,
847 model_part: Optional[str] = None,
848 model_part_options: Optional[dict] = None,
849 config_save_path: Optional[str] = None,
850 *,
851 extra_output_config: Optional[ExtraOutputConfig] = None,
852 **kwargs,
853) -> None:
854 """Quantize and compile through the ConfigManager-resolved pipeline.
855
856 Serves the backends in ``MODEL_DICT_BACKENDS``; the callers
857 (:func:`mxq_compile_from_source`, :func:`mxq_compile_from_mblt`) have
858 already validated ``target_device`` and normalized ``backend``.
859 """
860 request = build_quantize_request(
861 model=model,
862 save_path=save_path,
863 backend=backend,
864 target_device=target_device,
865 feed_dict=feed_dict,
866 dynamic_axes=dynamic_axes,
867 multi_shape=multi_shape,
868 yolo_decode_include=yolo_decode_include,
869 exclude_first_subgraph=exclude_first_subgraph,
870 save_subgraph_type=save_subgraph_type,
871 output_subgraph_path=output_subgraph_path,
872 calib_data_path=calib_data_path,
873 device=device,
874 inference_scheme=inference_scheme,
875 use_random_calib=use_random_calib,
876 cpu_offload=cpu_offload,
877 optimize_option=optimize_option,
878 buffer_mode=buffer_mode,
879 force_npu_input_reposition=force_npu_input_reposition,
880 force_npu_output_reposition=force_npu_output_reposition,
881 image_channels=image_channels,
882 split_blocks=split_blocks,
883 split_parts=split_parts,
884 bias_correction=bias_correction,
885 config_preset=config_preset,
886 compile_config=compile_config,
887 resource_management_config=resource_management_config,
888 calibration_config=calibration_config,
889 bit_config=bit_config,
890 llm_config=llm_config,
891 hessian_quant_config=hessian_quant_config,
892 bias_correction_config=bias_correction_config,
893 mod_config=mod_config,
894 equivalent_transformation_config=equivalent_transformation_config,
895 search_weight_scale_config=search_weight_scale_config,
896 save_sample_config=save_sample_config,
897 uint8_input_config=uint8_input_config,
898 preprocessing_config=preprocessing_config,
899 extra_output_config=extra_output_config,
900 model_part=model_part,
901 model_part_options=model_part_options,
902 **kwargs,
903 )
904 resolved_request = ConfigManager.resolve_quantize_request(request)
905 if config_save_path is not None:
906 # Written before the (potentially hours-long) compile starts, so the
907 # record survives a failure. Two fields still hold their defaults here
908 # because ``MxqCompileStage`` fills them at execution time:
909 # ``model_paths`` -- which for a raw-model input is a throwaway path in
910 # the pipeline tempdir -- and ``runtime_options.version``.
911 save_compile_config(resolved_request.compile_config, config_save_path)
912 execute_quantize_request(resolved_request)
913
914
915def _mblt_compile_pipeline(
916 model: str | object,
917 target_device: str,
918 mblt_save_path: str | list[str],
919 backend="onnx",
920 device: object = UNSET,
921 feed_dict=None,
922 dynamic_axes=None,
923 multi_shape: Optional[dict] = None,
924 yolo_decode_include=False,
925 cpu_offload: object = UNSET,
926 exclude_first_subgraph=False,
927 model_part: Optional[str] = None,
928 model_part_options: Optional[dict] = None,
929 config_preset: Optional[str] = None,
930 compile_config: Optional[CompileConfig | str] = None,
931 compilation_mode: Optional[str] = None,
932 **kwargs,
933) -> None:
934 """Export to ``.mblt`` through the ConfigManager-resolved pipeline.
935
936 ``compilation_mode`` is an internal knob: one of
937 ``{"release", "dev", "debug"}`` or ``None`` (default) to defer to the
938 ``MBLT_APP_ENV`` environment variable. When set it overrides the env
939 and drives the parser's ``inference_validation`` / ``device_alloc``
940 / ``log_level`` cascade — the same preset as ``MBLT_APP_ENV``. Not
941 surfaced in the CLI on purpose; intended for in-process scripts and
942 tests that want scoped dev-mode validation without mutating
943 process-wide environment.
944 """
945 request = build_mblt_compile_request(
946 model=model,
947 mblt_save_path=mblt_save_path,
948 backend=backend,
949 target_device=target_device,
950 device=device,
951 feed_dict=feed_dict,
952 dynamic_axes=dynamic_axes,
953 multi_shape=multi_shape,
954 yolo_decode_include=yolo_decode_include,
955 cpu_offload=cpu_offload,
956 exclude_first_subgraph=exclude_first_subgraph,
957 model_part=model_part,
958 model_part_options=model_part_options,
959 config_preset=config_preset,
960 compile_config=compile_config,
961 compilation_mode=compilation_mode,
962 **kwargs,
963 )
964 resolved_request = ConfigManager.resolve_mblt_compile_request(request)
965 execute_mblt_compile_request(resolved_request)
966
967
968def mblt_compile_with_callback(
969 model: str,
970 mblt_save_path: str,
971 target_device: str,
972 backend: str = "onnx",
973 device: Optional[str] = None,
974 cpu_offload: Optional[bool] = None,
975 config_preset: Optional[str] = None,
976 compile_config: Optional[CompileConfig | str] = None,
977 compilation_mode: Optional[str] = None,
978 *,
979 progress_callback: Callable[[int, str], None],
980) -> None:
981 """Compile-to-mblt entry point with progress callbacks.
982
983 Only forwards explicitly-set values so the ``UNSET`` defaults (and the
984 merged ``CompileConfig`` behind them) apply when the caller omits a flag --
985 which is why this goes through ``_mblt_compile_dispatch`` rather than
986 :func:`mblt_compile`, whose own ``device`` / ``cpu_offload`` defaults would
987 override the config. ``target_device`` is required and has no config
988 fallback — callers (e.g. the CLI) validate it via
989 ``validate_target_device`` before invoking (QC-228).
990 """
991 call_kwargs: dict[str, Any] = {
992 "model": model,
993 "mblt_save_path": mblt_save_path,
994 "backend": backend,
995 "target_device": target_device,
996 }
997 if device is not None:
998 call_kwargs["device"] = device
999 if cpu_offload is not None:
1000 call_kwargs["cpu_offload"] = cpu_offload
1001 if config_preset is not None:
1002 call_kwargs["config_preset"] = config_preset
1003 if compile_config is not None:
1004 call_kwargs["compile_config"] = compile_config
1005 if compilation_mode is not None:
1006 call_kwargs["compilation_mode"] = compilation_mode
1007
1008 with progress_context(progress_callback):
1009 emit_progress(0, "Initializing compiler", log=False)
1010 _mblt_compile_dispatch(**call_kwargs)
1011 emit_progress(100, "Complete", log=False)
1012
1013
1014def mxq_compile_with_callback(
1015 model: str,
1016 target_device: str,
1017 save_path: str,
1018 backend: str = "onnx",
1019 device: Optional[str] = None,
1020 calib_data_path: Optional[Union[str, List[str]]] = None,
1021 use_random_calib: Optional[bool] = None,
1022 config_preset: Optional[str] = None,
1023 compile_config: Optional[CompileConfig | str] = None,
1024 config_save_path: Optional[str] = None,
1025 *,
1026 progress_callback: Callable[[int, str], None],
1027) -> None:
1028 """Compile/quantize entry point with progress callbacks.
1029
1030 ``target_device`` is required and has no config fallback — callers
1031 (e.g. the CLI) validate it via ``validate_target_device`` before
1032 invoking, and the wrapper forwards it unconditionally (QC-228).
1033
1034 Like :func:`mblt_compile_with_callback`, only explicitly-set values are
1035 forwarded, so :func:`mxq_compile`'s ``UNSET`` defaults (and the merged
1036 ``CompileConfig`` behind them) still apply to whatever the caller omits.
1037 ``config_save_path`` passes straight through, and the resolved config is
1038 written there before compilation starts; this is what backs the CLI's
1039 ``--config-save-path`` on ``quantize`` / ``compile``.
1040 """
1041 call_kwargs: dict[str, object] = {
1042 "model": model,
1043 "save_path": save_path,
1044 "backend": backend,
1045 "target_device": target_device,
1046 }
1047 if device is not None:
1048 call_kwargs["device"] = device
1049 if calib_data_path is not None:
1050 call_kwargs["calib_data_path"] = calib_data_path
1051 if use_random_calib is not None:
1052 call_kwargs["use_random_calib"] = use_random_calib
1053 if config_preset is not None:
1054 call_kwargs["config_preset"] = config_preset
1055 if compile_config is not None:
1056 call_kwargs["compile_config"] = compile_config
1057 if config_save_path is not None:
1058 call_kwargs["config_save_path"] = config_save_path
1059
1060 with progress_context(progress_callback):
1061 emit_progress(0, "Initializing compiler", log=False)
1062 mxq_compile(**call_kwargs)
1063 emit_progress(100, "Complete", log=False)
1064
1065
1066
Auto-generated config module from config_schema.yaml.
Definition __init__.py:1