diff --git a/crates/core/parity_tests/perfmodel/test_engine_step_parity.py b/crates/core/parity_tests/perfmodel/test_engine_step_parity.py index 1cf6f4851..1bcf2f252 100644 --- a/crates/core/parity_tests/perfmodel/test_engine_step_parity.py +++ b/crates/core/parity_tests/perfmodel/test_engine_step_parity.py @@ -789,7 +789,10 @@ def __call__(self): def _quiet_call(func, *args, **kwargs): """Keep interpolation loader chatter out of parity test output.""" - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): + with ( + contextlib.redirect_stdout(io.StringIO()), + contextlib.redirect_stderr(io.StringIO()), + ): return func(*args, **kwargs) @@ -832,12 +835,24 @@ def _static_metrics( "transfer_policy": case.transfer_policy, "moe_quant_mode": case.moe_quant_mode, } - ctx_result = _MemoizedCall(lambda: _quiet_call(cli_estimate, mode="static_ctx", **kwargs)) - gen_result = _MemoizedCall(lambda: _quiet_call(cli_estimate, mode="static_gen", **kwargs)) - context_ms = _safe_value(lambda: ctx_result().summary.get_summary_df().iloc[0]["context_latency"]) - generation_ms = _safe_value(lambda: gen_result().summary.get_summary_df().iloc[0]["generation_latency"]) - if isinstance(context_ms, _ErrorSentinel) or isinstance(generation_ms, _ErrorSentinel): - total: float | _ErrorSentinel = context_ms if isinstance(context_ms, _ErrorSentinel) else generation_ms + ctx_result = _MemoizedCall( + lambda: _quiet_call(cli_estimate, mode="static_ctx", **kwargs) + ) + gen_result = _MemoizedCall( + lambda: _quiet_call(cli_estimate, mode="static_gen", **kwargs) + ) + context_ms = _safe_value( + lambda: ctx_result().summary.get_summary_df().iloc[0]["context_latency"] + ) + generation_ms = _safe_value( + lambda: gen_result().summary.get_summary_df().iloc[0]["generation_latency"] + ) + if isinstance(context_ms, _ErrorSentinel) or isinstance( + generation_ms, _ErrorSentinel + ): + total: float | _ErrorSentinel = ( + context_ms if isinstance(context_ms, _ErrorSentinel) else generation_ms + ) else: total = context_ms + generation_ms metrics: dict[str, float | _ErrorSentinel] = { @@ -855,8 +870,12 @@ def _static_metrics( metrics["gen_energy_wms"] = _safe_value( lambda: sum(gen_result().summary.get_generation_energy_wms_dict().values()) ) - metrics["ctx_power_w"] = _safe_value(lambda: ctx_result().summary.get_context_power_avg()) - metrics["gen_power_w"] = _safe_value(lambda: gen_result().summary.get_generation_power_avg()) + metrics["ctx_power_w"] = _safe_value( + lambda: ctx_result().summary.get_context_power_avg() + ) + metrics["gen_power_w"] = _safe_value( + lambda: gen_result().summary.get_generation_power_avg() + ) return metrics @@ -1024,7 +1043,12 @@ def _case_database(case: EngineStepParityCase): defaults, or a mode/policy-configured query view for HYBRID/EMPIRICAL cases (mirrors what `cli_estimate` builds internally).""" if case.database_mode == "SILICON" and case.transfer_policy is None: - return _quiet_call(perf_database.get_database, case.system_name, case.backend_name, case.backend_version) + return _quiet_call( + perf_database.get_database, + case.system_name, + case.backend_name, + case.backend_version, + ) return _quiet_call( perf_database.get_database_view, case.system_name, @@ -1046,7 +1070,9 @@ def _case_model_config(case: EngineStepParityCase) -> config.ModelConfig: moe_tp_size=case.moe_tp_size, moe_ep_size=case.moe_ep_size, cp_size=case.cp_size, - moe_quant_mode=(common.MoEQuantMode[case.moe_quant_mode] if case.moe_quant_mode else None), + moe_quant_mode=( + common.MoEQuantMode[case.moe_quant_mode] if case.moe_quant_mode else None + ), nextn=case.nextn, ) @@ -1066,7 +1092,9 @@ def _cp_static_ctx_ms(case: EngineStepParityCase) -> float: raise RuntimeError( f"failed to load perf database for {case.system_name}/{case.backend_name}/{case.backend_version}" ) - model = _quiet_call(get_model, case.model_path, _case_model_config(case), case.backend_name) + model = _quiet_call( + get_model, case.model_path, _case_model_config(case), case.backend_name + ) backend = get_backend(case.backend_name) runtime_config = config.RuntimeConfig( batch_size=case.batch_size, @@ -1089,7 +1117,9 @@ def _rust_mixed_step_ms(case: EngineStepParityCase) -> float: raise RuntimeError( f"failed to load perf database for {case.system_name}/{case.backend_name}/{case.backend_version}" ) - model = _quiet_call(get_model, case.model_path, _case_model_config(case), case.backend_name) + model = _quiet_call( + get_model, case.model_path, _case_model_config(case), case.backend_name + ) shape = _mix_step_shape(case) return rust_engine_step.estimate_mixed_step_latency_with_rust( model, @@ -1134,14 +1164,18 @@ def _parity_mismatch_reason( ) continue # Both errored with the same kind — symmetric. Pass. - rows.append(f"{name:<{metric_width}} {'ERROR':>10} {'ERROR':>10} {'-':>10} {'-':>10} {'-':>10} sym") + rows.append( + f"{name:<{metric_width}} {'ERROR':>10} {'ERROR':>10} {'-':>10} {'-':>10} {'-':>10} sym" + ) continue if py_err != rs_err: # Asymmetric — one errored, the other didn't. has_mismatch = True py_repr = repr(python_value) if py_err else f"{python_value:.3f}" rs_repr = repr(rust_value) if rs_err else f"{rust_value:.3f}" - rows.append(f"{name:<{metric_width}} {py_repr:>10} {rs_repr:>10} {'-':>10} {'-':>10} {'-':>10} asym") + rows.append( + f"{name:<{metric_width}} {py_repr:>10} {rs_repr:>10} {'-':>10} {'-':>10} {'-':>10} asym" + ) continue # Both compute — apply numeric tolerance. allowed = max(abs(python_value) * rtol, 1e-9) @@ -1238,7 +1272,9 @@ def _comparison_metrics( return {name: (python_metrics[name], rust_metrics[name]) for name in rust_metrics} -def _static_comparison_metrics(case: EngineStepParityCase) -> dict[str, tuple[float, float]]: +def _static_comparison_metrics( + case: EngineStepParityCase, +) -> dict[str, tuple[float, float]]: return _comparison_metrics(case, "static") @@ -1248,11 +1284,15 @@ def _mixed_step_comparison_metrics( return _comparison_metrics(case, "mixed") -def _agg_comparison_metrics(case: EngineStepParityCase) -> dict[str, tuple[float, float]]: +def _agg_comparison_metrics( + case: EngineStepParityCase, +) -> dict[str, tuple[float, float]]: return _comparison_metrics(case, "agg") -def _disagg_comparison_metrics(case: EngineStepParityCase) -> dict[str, tuple[float, float]]: +def _disagg_comparison_metrics( + case: EngineStepParityCase, +) -> dict[str, tuple[float, float]]: return _comparison_metrics(case, "disagg") @@ -1280,7 +1320,9 @@ def load_parity_golden(filename: str) -> dict: if cached is None: path = GOLDEN_DIR / filename if not path.is_file(): - pytest.fail(f"missing golden fixture {path}; {_REGENERATE_HINT}", pytrace=False) + pytest.fail( + f"missing golden fixture {path}; {_REGENERATE_HINT}", pytrace=False + ) cached = _GOLDEN_CACHE[filename] = json.loads(path.read_text()) return cached @@ -1310,7 +1352,9 @@ def _golden_python_metrics( metric-set change and needs regeneration. """ case_id = _case_golden_id(case) - record = load_parity_golden("engine_step.json")["cases"].get(case_id, {}).get(surface) + record = ( + load_parity_golden("engine_step.json")["cases"].get(case_id, {}).get(surface) + ) if record is None: pytest.fail( f"no engine-step golden for case '{case_id}' surface '{surface}'; {_REGENERATE_HINT}", @@ -1327,7 +1371,11 @@ def _golden_python_metrics( pytrace=False, ) return { - name: (_ErrorSentinel.from_kind(value["error"]) if isinstance(value, dict) else float(value)) + name: ( + _ErrorSentinel.from_kind(value["error"]) + if isinstance(value, dict) + else float(value) + ) for name, value in values.items() } @@ -1578,15 +1626,26 @@ class TestRustEngineHandleDatabasePolicyIdentity: ANCHOR_MS = 42.4307555484161 # the issue #1498 adjudicated static_ctx sum def _static_ctx_ms(self, model, view) -> float: - rc = config.RuntimeConfig(batch_size=1, beam_width=1, isl=8192, osl=8, prefix=0, engine_step_backend="rust") - ctx_latency, _gen, *_ = rust_engine_step.estimate_static_latency_breakdown_with_rust( - model, view, rc, "static_ctx", 1, 1.0 + rc = config.RuntimeConfig( + batch_size=1, + beam_width=1, + isl=8192, + osl=8, + prefix=0, + engine_step_backend="rust", + ) + ctx_latency, _gen, *_ = ( + rust_engine_step.estimate_static_latency_breakdown_with_rust( + model, view, rc, "static_ctx", 1, 1.0 + ) ) return float(sum(ctx_latency.values())) def _build(self): (case,) = DSV4_CP_CASES[0].values - model = _quiet_call(get_model, case.model_path, _case_model_config(case), case.backend_name) + model = _quiet_call( + get_model, case.model_path, _case_model_config(case), case.backend_name + ) off = _quiet_call( perf_database.get_database_view, case.system_name, @@ -1605,14 +1664,18 @@ def _build(self): pytest.skip("no perf database for the DSV4 CP case identity") return model, off, on - def test_off_warmed_cache_does_not_fail_the_shared_on_view(self, monkeypatch: pytest.MonkeyPatch) -> None: + def test_off_warmed_cache_does_not_fail_the_shared_on_view( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: _prepare_rust_core(monkeypatch) # the ONLY cache clear in this ordering model, off, on = self._build() with pytest.raises(errors.PerfDataNotAvailableError): self._static_ctx_ms(model, off) assert self._static_ctx_ms(model, on) == pytest.approx(self.ANCHOR_MS, rel=1e-9) - def test_on_warmed_cache_does_not_answer_for_the_shared_off_view(self, monkeypatch: pytest.MonkeyPatch) -> None: + def test_on_warmed_cache_does_not_answer_for_the_shared_off_view( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: _prepare_rust_core(monkeypatch) # the ONLY cache clear in this ordering model, off, on = self._build() assert self._static_ctx_ms(model, on) == pytest.approx(self.ANCHOR_MS, rel=1e-9) @@ -1828,7 +1891,9 @@ def test_hybrid_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_static_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _static_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -1841,7 +1906,9 @@ def test_hybrid_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_mixed_step_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _mixed_step_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -1854,7 +1921,9 @@ def test_hybrid_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_agg_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _agg_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -1867,7 +1936,9 @@ def test_hybrid_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_disagg_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _disagg_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -1922,7 +1993,9 @@ def test_sol_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_static_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _static_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -1935,7 +2008,9 @@ def test_sol_parity( ) -> None: _prepare_rust_core(monkeypatch) - reason = _parity_mismatch_reason(_mixed_step_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL) + reason = _parity_mismatch_reason( + _mixed_step_comparison_metrics(case), rtol=HYBRID_PARITY_RTOL + ) assert reason is None, reason @@ -2011,7 +2086,9 @@ def _build_case_golden_ids() -> dict[EngineStepParityCase, str]: (case,) = param.values existing = mapping.get(case) if existing is not None and existing != param.id: - raise AssertionError(f"case {case!r} carries two golden ids: {existing!r} / {param.id!r}") + raise AssertionError( + f"case {case!r} carries two golden ids: {existing!r} / {param.id!r}" + ) mapping[case] = param.id return mapping @@ -2049,7 +2126,9 @@ def test_golden_comparison_detects_drift( monkeypatch.setattr( sys.modules[__name__], "load_parity_golden", - lambda filename: doctored if filename == "engine_step.json" else original(filename), + lambda filename: doctored + if filename == "engine_step.json" + else original(filename), ) reason = _parity_mismatch_reason(_static_comparison_metrics(case)) assert reason is not None, "5% golden drift on static_ctx was not detected" @@ -2073,10 +2152,14 @@ def test_golden_error_asymmetry_detected( monkeypatch.setattr( sys.modules[__name__], "load_parity_golden", - lambda filename: doctored if filename == "engine_step.json" else original(filename), + lambda filename: doctored + if filename == "engine_step.json" + else original(filename), ) reason = _parity_mismatch_reason(_static_comparison_metrics(case)) - assert reason is not None, "golden-error vs rust-value asymmetry was not detected" + assert reason is not None, ( + "golden-error vs rust-value asymmetry was not detected" + ) assert "asym" in reason, reason @@ -2084,7 +2167,9 @@ def _rust_static_breakdown(case: EngineStepParityCase): """Drive the rust engine-step bridge directly (no cli_estimate error wrapping) so the exception object crossing the FFI is what the test sees.""" database = _case_database(case) - model = _quiet_call(get_model, case.model_path, _case_model_config(case), case.backend_name) + model = _quiet_call( + get_model, case.model_path, _case_model_config(case), case.backend_name + ) runtime_config = config.RuntimeConfig( batch_size=case.batch_size, beam_width=1, @@ -2168,7 +2253,9 @@ def test_silicon_missing_dtype_on_lazy_family_matches_python( # missing-dtype on an eager op) left uncovered until the flops # resolution was hoisted before every load/key lookup. _prepare_rust_core(monkeypatch) - case = EngineStepParityCase(model_path="nvidia/MiniMax-M2.5-NVFP4", system_name="h200_sxm") + case = EngineStepParityCase( + model_path="nvidia/MiniMax-M2.5-NVFP4", system_name="h200_sxm" + ) with pytest.raises(errors.MissingSystemFlopsError) as excinfo: _rust_static_breakdown(case) assert "missing system flops" in str(excinfo.value), str(excinfo.value) @@ -2197,7 +2284,9 @@ def test_pre_sm89_fp8_kv_generation_returns_shipped_silicon(self) -> None: # (generation_attn_mode) must keep those rows queryable instead of # demanding an fp8_tc_flops entry a100 must never define (the # support-matrix FP8 gate is keyed on that entry's presence). - database = _quiet_call(perf_database.get_database, "a100_sxm", "trtllm", "1.0.0") + database = _quiet_call( + perf_database.get_database, "a100_sxm", "trtllm", "1.0.0" + ) from aiconfigurator_core.sdk.engine import _evaluate_single_op from aiconfigurator_core.sdk.operations.mla import GenerationMLA @@ -2205,7 +2294,9 @@ def _gen_mla(kv_mode): # The retired query_generation_mla shim's exact twin (the # database's live SILICON view). op = GenerationMLA("generation_mla_query", 1.0, 64, kv_mode) - return _evaluate_single_op(database, op, is_context=False, batch_size=1, s=65) + return _evaluate_single_op( + database, op, is_context=False, batch_size=1, s=65 + ) fp8_kv = _gen_mla(common.KVCacheQuantMode.fp8) bf16_kv = _gen_mla(common.KVCacheQuantMode.bfloat16) @@ -2230,7 +2321,9 @@ def test_hybrid_xop_run_records_tier( # tier. Python probes record {xop, xshape}; the rust path must land on # the same worst_provenance. _prepare_rust_core(monkeypatch) - case = EngineStepParityCase(model_path="MiniMaxAI/MiniMax-M3", database_mode="HYBRID") + case = EngineStepParityCase( + model_path="MiniMaxAI/MiniMax-M3", database_mode="HYBRID" + ) with util_empirical.capture_provenance() as tags: metrics = _static_metrics(case) assert not isinstance(metrics["total_ms"], _ErrorSentinel), repr(metrics) @@ -2430,15 +2523,23 @@ def _build(self): forward_model="fpm", ) model = get_model(_FPM_MODEL, cfg, "vllm") - database = _quiet_call(perf_database.get_database, "b200_sxm", "vllm", _FPM_VERSION) + database = _quiet_call( + perf_database.get_database, "b200_sxm", "vllm", _FPM_VERSION + ) return model, get_backend("vllm"), database def _static(self, model, backend, database, mode, batch, isl, osl, prefix): - rc = config.RuntimeConfig(batch_size=batch, beam_width=1, isl=isl, osl=osl, prefix=prefix) + rc = config.RuntimeConfig( + batch_size=batch, beam_width=1, isl=isl, osl=osl, prefix=prefix + ) def thunk(): summary = backend.run_static(model, database, rc, mode=mode) - d = summary.get_context_latency_dict() if mode == "static_ctx" else summary.get_generation_latency_dict() + d = ( + summary.get_context_latency_dict() + if mode == "static_ctx" + else summary.get_generation_latency_dict() + ) return sum(d.values()) return _safe_value(thunk) @@ -2450,12 +2551,16 @@ def _assert_frozen(frozen, rs, point): f"{point}: expected symmetric {frozen}, got {rs!r}" ) return - assert not isinstance(rs, _ErrorSentinel), f"{point}: frozen={frozen} but live raised {rs!r}" + assert not isinstance(rs, _ErrorSentinel), ( + f"{point}: frozen={frozen} but live raised {rs!r}" + ) allowed = max(abs(frozen) * PARITY_RTOL, 1e-9) - assert abs(rs - frozen) <= allowed, f"{point}: frozen={frozen} rs={rs} delta={abs(rs - frozen)}" + assert abs(rs - frozen) <= allowed, ( + f"{point}: frozen={frozen} rs={rs} delta={abs(rs - frozen)}" + ) def test_fpm_arena_selects_the_fpm_engine(self, fpm_systems_root, monkeypatch): - # Review finding (#1461): from_native() dropped forward_model, so the + # Review finding (#1461): the former constructor dropped forward_model, so the # FPM arena always compiled the op_level engine. A decode-only # estimate hitting the fpm_forward table's exact row proves the # whole-model engine was selected through the supported predictor API. @@ -2463,26 +2568,25 @@ def test_fpm_arena_selects_the_fpm_engine(self, fpm_systems_root, monkeypatch): from aiconfigurator_core.sdk.rust_engine_step import RustForwardPassPerfModel config = { - "schema_version": 1, - "model_name": _FPM_MODEL, - "system_name": "b200_sxm", + "model": _FPM_MODEL, + "system": "b200_sxm", "backend": "vllm", "backend_version": _FPM_VERSION, - "systems_path": str(fpm_systems_root), - "tp_size": 2, - "pp_size": 1, - "attention_dp_size": 1, + "systems_paths": [str(fpm_systems_root)], + "tp": 2, + "pp": 1, + "attention_dp": 1, "moe_tp_size": 1, "moe_ep_size": 2, - "weight_dtype": "fp8_block", - "moe_dtype": "fp8_block", - "activation_dtype": "bfloat16", - "kv_cache_dtype": "fp8", + "gemm_quant_mode": "fp8_block", + "moe_quant_mode": "fp8_block", + "fmha_quant_mode": "bfloat16", + "kvcache_quant_mode": "fp8", "kv_block_size": None, - "nextn": None, "forward_model": "fpm", + "fallback_policy": "error", } - model = RustForwardPassPerfModel.from_native(config) + model = RustForwardPassPerfModel.best_available(config) decode_only = [ { "version": 1, @@ -2536,11 +2640,19 @@ def test_fpm_spec_tags(self, fpm_systems_root, monkeypatch): ("static_gen", 4, 1024, 2, 0), # exact decode hit at B=4 ("static_gen", 2, 1024, 2, 0), # uncollected batch -> transfer (SOL) ("static_gen", 4, 9_000_000, 2, 0), # out of domain -> both error - ("static_ctx", 16, 256, 1, 0), # above the batch ceiling -> pure clamp (kv/T = 0) + ( + "static_ctx", + 16, + 256, + 1, + 0, + ), # above the batch ceiling -> pure clamp (kv/T = 0) ("static_ctx", 16, 320, 1, 256), # high KV pressure -> SOL-rescaled clamp ], ) - def test_fpm_static_parity(self, fpm_systems_root, monkeypatch, mode, batch, isl, osl, prefix): + def test_fpm_static_parity( + self, fpm_systems_root, monkeypatch, mode, batch, isl, osl, prefix + ): _prepare_rust_core(monkeypatch) model, backend, database = self._build() rs = self._static(model, backend, database, mode, batch, isl, osl, prefix) @@ -2555,16 +2667,27 @@ def test_fpm_static_parity(self, fpm_systems_root, monkeypatch, mode, batch, isl (0, 4, 1024, 2), # gen-only keeps full decode (0, 600, 100, 2), # gen-only across the decode regime boundary (eager side) (1024, 0, 1024, 2), # prefill-only chunk - (4096, 0, 256, 1), # 16 whole prefills: certified batch clamp to the ceiling + ( + 4096, + 0, + 256, + 1, + ), # 16 whole prefills: certified batch clamp to the ceiling ], ) - def test_fpm_mixed_step_parity(self, fpm_systems_root, monkeypatch, ctx_tokens, gen_tokens, isl, osl): + def test_fpm_mixed_step_parity( + self, fpm_systems_root, monkeypatch, ctx_tokens, gen_tokens, isl, osl + ): _prepare_rust_core(monkeypatch) model, backend, database = self._build() - rc = config.RuntimeConfig(batch_size=1, beam_width=1, isl=isl, osl=osl, prefix=0) + rc = config.RuntimeConfig( + batch_size=1, beam_width=1, isl=isl, osl=osl, prefix=0 + ) rs = _safe_value( - lambda: backend._get_mix_step_latency(model, database, rc, ctx_tokens, gen_tokens, isl, osl, 0)[0] + lambda: backend._get_mix_step_latency( + model, database, rc, ctx_tokens, gen_tokens, isl, osl, 0 + )[0] ) frozen = _FPM_MIXED_FROZEN[(ctx_tokens, gen_tokens, isl, osl)] self._assert_frozen(frozen, rs, f"mixed ctx={ctx_tokens} gen={gen_tokens}") @@ -2574,5 +2697,9 @@ def test_fpm_genonly_step_parity(self, fpm_systems_root, monkeypatch): model, backend, database = self._build() rc = config.RuntimeConfig(batch_size=1, beam_width=1, isl=1024, osl=2, prefix=0) - rs = _safe_value(lambda: backend._get_genonly_step_latency(model, database, rc, 4, 1023, 2)[0]) + rs = _safe_value( + lambda: backend._get_genonly_step_latency(model, database, rc, 4, 1023, 2)[ + 0 + ] + ) self._assert_frozen(_FPM_GENONLY_FROZEN, rs, "genonly gen=4") diff --git a/crates/core/src/lib.rs b/crates/core/src/lib.rs index 1a5bfe483..f9d8cfb7a 100644 --- a/crates/core/src/lib.rs +++ b/crates/core/src/lib.rs @@ -33,15 +33,24 @@ pub use replay::{ReplayReport, ReplaySpec, Replayer}; pub use perfmodel::EngineConfig; pub use perfmodel::{ AicError, BackendKind, DataType, ENGINE_CONFIG_SCHEMA_VERSION, ENGINE_SPEC_SCHEMA_VERSION, - EstimateSource, FPM_VERSION, ForwardPassMetrics, ForwardPassPerfDiagnostics, - ForwardPassPerfModel, ForwardPassPerfOptions, ForwardPassPerfReadiness, ForwardPassPerfSource, - KvCacheEstimate, KvCacheEstimateAdjusted, KvCacheEstimateError, KvCacheEstimateOptions, - KvCacheEstimateRequest, KvCacheMemoryFraction, MemoryBreakdown, ParallelMapping, - QuantizationConfig, QueuedRequestMetrics, ScheduledRequestMetrics, SpeculativeConfig, + EstimateSource, FPM_VERSION, ForwardPassFallbackPolicy, ForwardPassMetrics, + ForwardPassModelKind, ForwardPassPerfDiagnostics, ForwardPassPerfModel, + ForwardPassPerfModelConfig, ForwardPassPerfOptions, ForwardPassPerfProvenance, + ForwardPassPerfReadiness, ForwardPassPerfSource, KvCacheEstimate, KvCacheEstimateAdjusted, + KvCacheEstimateError, KvCacheEstimateOptions, KvCacheEstimateRequest, KvCacheMemoryFraction, + MemoryBreakdown, ParallelMapping, QuantizationConfig, QueuedRequestMetrics, + ScheduledRequestMetrics, SpeculativeConfig, }; #[cfg(feature = "python")] -pub use perfmodel::{AicEngine, AicEngineBuilder, estimate_kv_cache}; +pub use perfmodel::{ + AicEngine, + // Low-level Rust embedder API for compiled step-latency handles. This is + // deliberately not registered on the Python module and is not an + // alternative ForwardPassPerfModel construction boundary. + AicEngineBuilder, + estimate_kv_cache, +}; // The imported perf-model sources historically used additional crate-root // module paths. Keep these module aliases crate-private so the mirror subtree diff --git a/crates/core/src/perfmodel/engine/runtime.rs b/crates/core/src/perfmodel/engine/runtime.rs index f56286360..539b6035a 100644 --- a/crates/core/src/perfmodel/engine/runtime.rs +++ b/crates/core/src/perfmodel/engine/runtime.rs @@ -301,6 +301,24 @@ impl Engine { } } + /// Eagerly validate the whole-forward database cell selected by an FPM + /// engine. The table itself is lazy, so compiling the engine is not enough + /// to prove that the requested model identity has collected prefill and + /// decode data. The canonical forward-pass constructor calls this before + /// returning so a missing/mismatched FPM pair fails before search starts. + pub(crate) fn validate_forward_pass_readiness(&self) -> Result<(), AicError> { + let Some((prefill, decode)) = self.fpm_ops() else { + return Ok(()); + }; + self.db + .fpm_forward + .select_cell(&prefill.match_identity, &prefill.model_path)?; + self.db + .fpm_forward + .select_cell(&decode.match_identity, &decode.model_path)?; + Ok(()) + } + /// Convenience constructor: deserialize a bincode `EngineSpec` and load the /// matching `PerfDatabase` from its identity, then [`Engine::build`]. /// @@ -344,12 +362,10 @@ impl Engine { DatabaseMode::Silicon | DatabaseMode::Hybrid )), spec.engine.strict_provenance, - // Estimate-only systems (a spec yaml with no collected data) may - // back a SOL view: every SOL answer is analytic from the system - // spec, so tolerate a missing perf-data directory under SOL and - // let table-backed lookups miss lazily. All other modes keep the - // loud load-time gate. - spec.engine.database_mode == DatabaseMode::Sol, + // Formula/fallback-capable modes can serve a system spec without a + // collected primary directory. SILICON alone requires exact data + // at load time; every other mode owns its typed miss behavior. + spec.engine.database_mode != DatabaseMode::Silicon, )? .with_mode(spec.engine.database_mode, transfer_policy); Engine::build(spec, Arc::new(db)) @@ -1717,6 +1733,21 @@ mod tests { assert!(err.to_string().contains("exactly one FpmForward"), "{err}"); } + #[test] + fn fpm_readiness_eagerly_validates_the_exact_model_cell() { + let tmp = tempfile::tempdir().unwrap(); + let mut engine = build_fpm_engine(tmp.path(), None).unwrap(); + engine.validate_forward_pass_readiness().unwrap(); + + let [Op::FpmForward(prefill)] = engine.context_ops.as_mut_slice() else { + panic!("FPM fixture must contain one prefill op"); + }; + prefill.model_path = "org/uncollected-model".into(); + let err = engine.validate_forward_pass_readiness().unwrap_err(); + assert!(err.to_string().contains("No FPM cell matches"), "{err}"); + assert!(err.to_string().contains("uncollected-model"), "{err}"); + } + /// The marginal-decode mixed composition, exact arithmetic over the /// fixture rows: the prefill component prices the step's SCHEDULED TOTAL /// (ctx + gen tokens) on the prefill curve; decode is the in-curve lerp diff --git a/crates/core/src/perfmodel/fpm/config.rs b/crates/core/src/perfmodel/fpm/config.rs new file mode 100644 index 000000000..ec0efa7d3 --- /dev/null +++ b/crates/core/src/perfmodel/fpm/config.rs @@ -0,0 +1,227 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Canonical construction contract for [`super::ForwardPassPerfModel`]. + +use std::path::PathBuf; + +use serde::{Deserialize, Serialize}; + +use crate::common::enums::{DatabaseMode, TransferPolicy}; +use crate::{AicError, BackendKind}; + +const fn one() -> u32 { + 1 +} + +/// Forward-pass modeling implementation selected for the compiled engine. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ForwardPassModelKind { + /// Compose the forward pass from individually modeled operations. + #[default] + OpLevel, + /// Query the exact whole-forward performance database. + Fpm, +} + +impl ForwardPassModelKind { + pub(crate) fn as_str(self) -> &'static str { + match self { + Self::OpLevel => "op_level", + Self::Fpm => "fpm", + } + } +} + +/// Policy used when the native AIC estimator cannot serve the requested config. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ForwardPassFallbackPolicy { + /// Fail closed. Replay and Sweeper use this policy unless explicitly changed. + #[default] + Error, + /// Use the in-memory regression model and require observations before estimates. + Regression, +} + +/// Immutable model identity and selection policy for a forward-pass estimator. +/// +/// This is the one public construction schema shared by Rust, Python, Replay, +/// Sweeper, and Planner. Runtime learning/tuning controls deliberately live in +/// [`super::ForwardPassPerfOptions`]. +#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ForwardPassPerfModelConfig { + pub model: String, + pub system: String, + pub backend: BackendKind, + #[serde(default)] + pub backend_version: Option, + + #[serde(default = "one", alias = "tp_size")] + pub tp: u32, + #[serde(default = "one", alias = "pp_size")] + pub pp: u32, + #[serde(default = "one", alias = "attention_dp_size")] + pub attention_dp: u32, + #[serde(default)] + pub moe_tp_size: Option, + #[serde(default)] + pub moe_ep_size: Option, + + #[serde(default, alias = "gemm_dtype")] + pub gemm_quant_mode: Option, + #[serde(default, alias = "moe_dtype")] + pub moe_quant_mode: Option, + #[serde(default, alias = "fmha_dtype")] + pub fmha_quant_mode: Option, + #[serde(default, alias = "kv_cache_dtype")] + pub kvcache_quant_mode: Option, + #[serde(default, alias = "comm_dtype")] + pub comm_quant_mode: Option, + + #[serde(default)] + pub nextn: u32, + #[serde(default)] + pub kv_block_size: Option, + #[serde(default)] + pub forward_model: ForwardPassModelKind, + #[serde(default)] + pub database_mode: DatabaseMode, + /// Explicit transfer-kind tokens. `None` means the core default (all). + #[serde(default)] + pub transfer_policy: Option>, + /// Ordered request-scoped systems roots. Empty uses normal package/env discovery. + #[serde(default)] + pub systems_paths: Vec, + #[serde(default)] + pub fallback_policy: ForwardPassFallbackPolicy, +} + +impl ForwardPassPerfModelConfig { + pub(crate) fn validate(&self) -> Result<(), AicError> { + if self.model.trim().is_empty() { + return Err(invalid_config("model cannot be empty")); + } + if self.system.trim().is_empty() { + return Err(invalid_config("system cannot be empty")); + } + if self + .backend_version + .as_ref() + .is_some_and(|value| value.trim().is_empty()) + { + return Err(invalid_config("backend_version cannot be empty")); + } + if self.tp == 0 || self.pp == 0 || self.attention_dp == 0 { + return Err(invalid_config("tp, pp, and attention_dp must be positive")); + } + if self.moe_tp_size.is_some() != self.moe_ep_size.is_some() { + return Err(invalid_config( + "moe_tp_size and moe_ep_size must be configured together", + )); + } + if let (Some(moe_tp), Some(moe_ep)) = (self.moe_tp_size, self.moe_ep_size) { + if moe_tp == 0 || moe_ep == 0 { + return Err(invalid_config( + "moe_tp_size and moe_ep_size must be positive", + )); + } + if u64::from(self.tp) * u64::from(self.attention_dp) + != u64::from(moe_tp) * u64::from(moe_ep) + { + return Err(invalid_config( + "topology requires tp * attention_dp == moe_tp_size * moe_ep_size", + )); + } + } + if self.nextn > 5 { + return Err(invalid_config("nextn must be in 0..=5")); + } + if self.forward_model == ForwardPassModelKind::Fpm && self.nextn != 0 { + return Err(invalid_config( + "forward_model='fpm' does not support MTP speculative decoding", + )); + } + TransferPolicy::from_wire(self.transfer_policy.as_deref()).map_err(invalid_config)?; + for root in &self.systems_paths { + if !root.is_dir() { + return Err(invalid_config(format!( + "systems_paths entry is not an existing directory: {}", + root.display() + ))); + } + } + Ok(()) + } +} + +fn invalid_config(message: impl Into) -> AicError { + AicError::InvalidEngineConfig(format!( + "invalid forward pass perf model config: {}", + message.into() + )) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn config() -> ForwardPassPerfModelConfig { + ForwardPassPerfModelConfig { + model: "Qwen/Qwen3-32B".into(), + system: "h200_sxm".into(), + backend: BackendKind::Vllm, + backend_version: Some("0.19.0".into()), + tp: 1, + pp: 1, + attention_dp: 1, + moe_tp_size: None, + moe_ep_size: None, + gemm_quant_mode: None, + moe_quant_mode: None, + fmha_quant_mode: None, + kvcache_quant_mode: None, + comm_quant_mode: None, + nextn: 0, + kv_block_size: None, + forward_model: ForwardPassModelKind::OpLevel, + database_mode: DatabaseMode::Silicon, + transfer_policy: None, + systems_paths: Vec::new(), + fallback_policy: ForwardPassFallbackPolicy::Error, + } + } + + #[test] + fn serde_defaults_are_fail_closed_and_typed() { + let parsed: ForwardPassPerfModelConfig = serde_json::from_value(serde_json::json!({ + "model": "Qwen/Qwen3-32B", + "system": "h200_sxm", + "backend": "vllm" + })) + .unwrap(); + assert_eq!(parsed.tp, 1); + assert_eq!(parsed.forward_model, ForwardPassModelKind::OpLevel); + assert_eq!(parsed.fallback_policy, ForwardPassFallbackPolicy::Error); + parsed.validate().unwrap(); + } + + #[test] + fn validation_rejects_invalid_policy_and_topology() { + let mut invalid = config(); + invalid.transfer_policy = Some(vec!["mystery".into()]); + assert!(invalid.validate().is_err()); + + invalid = config(); + invalid.moe_tp_size = Some(1); + invalid.moe_ep_size = Some(2); + assert!(invalid.validate().is_err()); + + invalid = config(); + invalid.forward_model = ForwardPassModelKind::Fpm; + invalid.nextn = 1; + assert!(invalid.validate().is_err()); + } +} diff --git a/crates/core/src/perfmodel/fpm/mod.rs b/crates/core/src/perfmodel/fpm/mod.rs index f1ae9af1d..9428d53f6 100644 --- a/crates/core/src/perfmodel/fpm/mod.rs +++ b/crates/core/src/perfmodel/fpm/mod.rs @@ -22,6 +22,7 @@ //! - [`samples`]: shared bucketed-sample infrastructure. //! - [`options`]: tuning controls. +mod config; mod correction; mod metrics; mod model; @@ -32,10 +33,11 @@ mod samples; #[cfg(test)] mod tests; +pub use config::{ForwardPassFallbackPolicy, ForwardPassModelKind, ForwardPassPerfModelConfig}; pub(crate) use metrics::validate_forward_pass_metrics; pub use metrics::{FPM_VERSION, ForwardPassMetrics, QueuedRequestMetrics, ScheduledRequestMetrics}; pub use model::{ - ForwardPassPerfDiagnostics, ForwardPassPerfModel, ForwardPassPerfReadiness, - ForwardPassPerfSource, + ForwardPassPerfDiagnostics, ForwardPassPerfModel, ForwardPassPerfProvenance, + ForwardPassPerfReadiness, ForwardPassPerfSource, }; pub use options::ForwardPassPerfOptions; diff --git a/crates/core/src/perfmodel/fpm/model.rs b/crates/core/src/perfmodel/fpm/model.rs index d839e2f89..2e23695b4 100644 --- a/crates/core/src/perfmodel/fpm/model.rs +++ b/crates/core/src/perfmodel/fpm/model.rs @@ -8,17 +8,17 @@ //! through [`crate::perfmodel::engine::Engine::forward_pass_time_ms`]. The online //! correction / regression / diagnostics / readiness logic is engine-agnostic. -#[cfg(feature = "python")] -use std::path::{Path, PathBuf}; +use std::path::PathBuf; use std::sync::Arc; use serde::{Deserialize, Serialize}; -#[cfg(feature = "python")] -use crate::perfmodel::EngineConfig; use crate::perfmodel::engine::Engine; use crate::{AicError, ForwardPassMetrics}; +#[cfg(feature = "python")] +use super::config::ForwardPassFallbackPolicy; +use super::config::ForwardPassPerfModelConfig; use super::correction::CorrectionBuckets; use super::metrics::validate_forward_pass_metrics; use super::options::{ForwardPassPerfOptions, validate_options}; @@ -41,6 +41,17 @@ pub struct ForwardPassPerfDiagnostics { pub correction_ready_buckets: usize, /// Fallback reason when `best_available` had to use regression instead of native AIC. pub last_warning: Option, + /// Exact immutable configuration and selected systems root used at construction. + pub provenance: Option, +} + +/// Resolved construction provenance pinned by Replay, Sweeper, and Planner. +#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)] +pub struct ForwardPassPerfProvenance { + /// Canonical config with the exact backend version and explicit transfer policy. + pub config: ForwardPassPerfModelConfig, + /// Root that supplied the native engine, or `None` for regression fallback. + pub selected_systems_root: Option, } /// Prediction backend currently used by `ForwardPassPerfModel`. @@ -109,6 +120,7 @@ pub struct ForwardPassPerfModel { mode: ForwardPassPerfMode, options: ForwardPassPerfOptions, last_warning: Option, + provenance: Option, } #[derive(Clone, Debug)] @@ -126,48 +138,10 @@ enum ForwardPassPerfMode { } impl ForwardPassPerfModel { - /// API: - /// `ForwardPassPerfModel::from_native(config, options) -> Result` - /// - /// Description: create a strict native AIC forward-pass model. - /// - /// Compiles `config` into an [`Engine`] by crossing into Python once - /// (mirroring [`crate::AicEngineBuilder`]): `compile_engine` walks the model - /// and returns bincoded spec bytes, then [`Engine::from_spec_bytes`] loads - /// the matching perf database. This constructor fails if `config` cannot be - /// compiled. Use `best_available` when unsupported native configs should - /// fall back to the learned regression model. - #[cfg(feature = "python")] - pub fn from_native( - config: EngineConfig, - options: ForwardPassPerfOptions, - ) -> Result { - validate_options(&options)?; - let engine = build_engine_via_python(&config, None)?; - Ok(Self::from_engine(Arc::new(engine), options)) - } - - /// API: - /// `ForwardPassPerfModel::from_native_with_roots(config, options, systems_root) -> Result` - /// - /// Description: create a strict native AIC forward-pass model with an - /// explicit `systems/` data root (forwarded to `compile_engine` and used to - /// load the perf database). Same tuning and failure behavior as - /// `from_native`. - #[cfg(feature = "python")] - pub fn from_native_with_roots( - config: EngineConfig, - options: ForwardPassPerfOptions, - systems_root: impl AsRef, - ) -> Result { - validate_options(&options)?; - let engine = build_engine_via_python(&config, Some(systems_root.as_ref()))?; - Ok(Self::from_engine(Arc::new(engine), options)) - } - /// Internal: build a native model directly from an already-compiled - /// [`Engine`]. Holds the actual native-mode logic; the public `from_native` - /// constructors compile the `Engine` (crossing into Python) and call this. + /// [`Engine`]. Holds the actual native-mode logic; the public + /// [`Self::best_available`] constructor compiles the `Engine` (crossing + /// into Python) and calls this. /// Used by the `#[cfg(test)]` suite to construct a native model from a /// hand-built fixture `Engine` without Python. pub(crate) fn from_engine(engine: Arc, options: ForwardPassPerfOptions) -> Self { @@ -178,6 +152,7 @@ impl ForwardPassPerfModel { }, options, last_warning: None, + provenance: None, } } @@ -190,7 +165,7 @@ impl ForwardPassPerfModel { /// `estimate_forward_pass_time_ms` for non-empty iterations until the /// inferred workload kind has at least `options.min_observations` tuning samples. /// Correction factor getters always return `None` in this mode. - pub fn from_regression(options: ForwardPassPerfOptions) -> Result { + pub(crate) fn from_regression(options: ForwardPassPerfOptions) -> Result { validate_options(&options)?; Ok(Self { mode: ForwardPassPerfMode::Regression { @@ -198,54 +173,73 @@ impl ForwardPassPerfModel { }, options, last_warning: None, + provenance: None, }) } /// API: /// `ForwardPassPerfModel::best_available(config, options) -> Result` /// - /// Description: create a native model when possible, otherwise fall back to - /// regression. + /// Description: create a native model when possible. Regression fallback + /// occurs only when `config.fallback_policy` explicitly requests it. /// /// Fallback reason is preserved in `diagnostics().last_warning`. The /// resulting model still uses the same FPM workload-kind inference and - /// tuning input contract as `from_native` and `from_regression`. + /// tuning input contract as native mode. #[cfg(feature = "python")] pub fn best_available( - config: EngineConfig, - options: ForwardPassPerfOptions, + config: ForwardPassPerfModelConfig, + options: Option, ) -> Result { - match Self::from_native(config, options.clone()) { - Ok(model) => Ok(model), - Err(err) if can_fallback_to_regression(&err) => { - Self::regression_with_warning(options, err) + config.validate()?; + let options = options.unwrap_or_default(); + validate_options(&options)?; + + let roots = resolve_systems_roots(&config)?; + let mut failures = Vec::new(); + for systems_root in roots { + match build_engine_via_python(&config, &systems_root) { + Ok(engine) => match engine.validate_forward_pass_readiness() { + Ok(()) => { + let mut resolved_config = config.clone(); + resolved_config.backend_version = Some(engine.database().version.clone()); + resolved_config.database_mode = engine.database().database_mode; + resolved_config.transfer_policy = + Some(transfer_policy_tokens(engine.database().transfer_policy)); + // Provenance is also the replayable construction input. Pin the + // root that actually supplied the engine instead of returning the + // caller's search list and requiring every consumer to rewrite it. + resolved_config.systems_paths = vec![systems_root.clone()]; + let provenance = ForwardPassPerfProvenance { + config: resolved_config, + selected_systems_root: Some(systems_root), + }; + let mut model = Self::from_engine(Arc::new(engine), options); + model.provenance = Some(provenance); + return Ok(model); + } + Err(err) if can_fallback_to_regression(&err) => failures.push(err), + Err(err) => return Err(err), + }, + Err(err) if can_fallback_to_regression(&err) => failures.push(err), + Err(err) => return Err(err), } - Err(err) => Err(err), } - } - /// API: - /// `ForwardPassPerfModel::best_available_with_roots(config, options, systems_root) -> Result` - /// - /// Description: create a `best_available` model with an explicit `systems/` - /// data root. - #[cfg(feature = "python")] - pub fn best_available_with_roots( - config: EngineConfig, - options: ForwardPassPerfOptions, - systems_root: impl AsRef, - ) -> Result { - match Self::from_native_with_roots(config, options.clone(), systems_root) { - Ok(model) => Ok(model), - Err(err) if can_fallback_to_regression(&err) => { - Self::regression_with_warning(options, err) + let err = failures.pop().unwrap_or_else(|| { + AicError::DataRoot("no systems root was available for forward-pass construction".into()) + }); + match config.fallback_policy { + ForwardPassFallbackPolicy::Regression => { + Self::regression_with_warning(config, options, err) } - Err(err) => Err(err), + ForwardPassFallbackPolicy::Error => Err(err), } } #[cfg(feature = "python")] fn regression_with_warning( + config: ForwardPassPerfModelConfig, options: ForwardPassPerfOptions, err: AicError, ) -> Result { @@ -253,6 +247,10 @@ impl ForwardPassPerfModel { model.last_warning = Some(format!( "native forward-pass estimator unavailable; using fallback regression: {err}" )); + model.provenance = Some(ForwardPassPerfProvenance { + config, + selected_systems_root: None, + }); Ok(model) } @@ -378,6 +376,7 @@ impl ForwardPassPerfModel { retained_observations: corrections.observation_count(), correction_ready_buckets: ready_buckets, last_warning: self.last_warning.clone(), + provenance: self.provenance.clone(), } } ForwardPassPerfMode::Regression { regressions } => { @@ -394,6 +393,7 @@ impl ForwardPassPerfModel { retained_observations: regressions.observation_count(), correction_ready_buckets: 0, last_warning: self.last_warning.clone(), + provenance: self.provenance.clone(), } } } @@ -452,6 +452,11 @@ impl ForwardPassPerfModel { &self.options } + /// Exact immutable construction identity and selected systems root. + pub fn provenance(&self) -> Option<&ForwardPassPerfProvenance> { + self.provenance.as_ref() + } + fn correction_factors(&self) -> Vec { match &self.mode { ForwardPassPerfMode::Native { corrections, .. } => corrections.correction_factors(), @@ -460,37 +465,43 @@ impl ForwardPassPerfModel { } } -/// Build a compiled [`Engine`] from an [`EngineConfig`] by crossing into Python -/// once to run `aiconfigurator.sdk.engine.compile_engine`, then loading the -/// matching perf database via [`Engine::from_spec_bytes`]. This is the internal -/// `EngineConfig` counterpart to [`crate::AicEngineBuilder`] and maps its -/// modular fields onto the flat `compile_engine` kwargs. -/// -/// `systems_root` overrides the bundled `systems/` dir for BOTH the -/// `compile_engine` call (`systems_path` kwarg) and the Rust-side perf-DB load. #[cfg(feature = "python")] fn build_engine_via_python( - config: &EngineConfig, - systems_root: Option<&Path>, + config: &ForwardPassPerfModelConfig, + systems_root: &std::path::Path, ) -> Result { - // `compile_engine`'s `systems_path` kwarg: explicit override -> config's - // own `systems_path` -> None (Python resolves it). - let systems_path: Option = systems_root - .map(PathBuf::from) - .or_else(|| config.systems_path.clone()); - // A non-UTF-8 override path cannot be passed through the Python kwarg; fail - // loudly rather than silently dropping the override. - let systems_path_str = match systems_path.as_ref() { - Some(p) => Some(p.to_str().ok_or_else(|| { - AicError::InvalidEngineConfig(format!( - "systems_path is not valid UTF-8: {}", - p.display() - )) - })?), - None => None, - }; - - crate::py::compile_engine_to_engine(config, systems_path_str) + let systems_path = systems_root.to_str().ok_or_else(|| { + AicError::InvalidEngineConfig(format!( + "systems_path is not valid UTF-8: {}", + systems_root.display() + )) + })?; + crate::py::compile_forward_pass_model_to_engine(config, systems_path) +} + +#[cfg(feature = "python")] +fn resolve_systems_roots(config: &ForwardPassPerfModelConfig) -> Result, AicError> { + if !config.systems_paths.is_empty() { + return Ok(config.systems_paths.clone()); + } + crate::py::resolve_systems_root(None) + .map(|root| vec![root]) + .map_err(|err| AicError::DataRoot(format!("resolve systems path: {err}"))) +} + +#[cfg(feature = "python")] +fn transfer_policy_tokens(policy: crate::common::enums::TransferPolicy) -> Vec { + use crate::common::enums::TransferKind; + [ + (TransferKind::XShape, "xshape"), + (TransferKind::XQuant, "xquant"), + (TransferKind::XProfile, "xprofile"), + (TransferKind::XOp, "xop"), + ] + .into_iter() + .filter(|(kind, _)| policy.contains(*kind)) + .map(|(_, token)| token.to_string()) + .collect() } #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] diff --git a/crates/core/src/perfmodel/fpm/options.rs b/crates/core/src/perfmodel/fpm/options.rs index 1e8fcde2f..a8fc5f0a9 100644 --- a/crates/core/src/perfmodel/fpm/options.rs +++ b/crates/core/src/perfmodel/fpm/options.rs @@ -24,6 +24,7 @@ pub(crate) const DEFAULT_MAX_KV_TOKENS: u32 = 2_000_000; /// observations before predicting from learned data, bucket observations by /// workload kind, and bound native correction factors to `[0.5, 2.0]`. #[derive(Clone, Debug, Serialize, Deserialize, PartialEq)] +#[serde(deny_unknown_fields)] pub struct ForwardPassPerfOptions { /// Maximum retained observations across all buckets for each inferred workload kind. #[serde(default = "default_max_observations")] diff --git a/crates/core/src/perfmodel/fpm/tests.rs b/crates/core/src/perfmodel/fpm/tests.rs index 931402073..fccd4fa28 100644 --- a/crates/core/src/perfmodel/fpm/tests.rs +++ b/crates/core/src/perfmodel/fpm/tests.rs @@ -359,6 +359,13 @@ fn options_default_directional_correction_factors() { assert_eq!(no_ceiling.max_slower_correction_factor, None); } +#[test] +fn options_reject_unknown_fields() { + let err = + serde_json::from_str::(r#"{"min_observation": 5}"#).unwrap_err(); + assert!(err.to_string().contains("unknown field"), "{err}"); +} + #[test] fn options_validate_directional_correction_factors() { let model = ForwardPassPerfModel::from_regression(ForwardPassPerfOptions { diff --git a/crates/core/src/perfmodel/mod.rs b/crates/core/src/perfmodel/mod.rs index 0842ece93..25fb55981 100644 --- a/crates/core/src/perfmodel/mod.rs +++ b/crates/core/src/perfmodel/mod.rs @@ -7,8 +7,11 @@ //! `compile_engine` walks the model once and emits an [`engine::spec::EngineSpec`] //! (op lists + [`EngineConfig`] identity); the Rust [`engine::Engine`] executes //! it without re-entering Python. With the `python` feature enabled, -//! [`AicEngineBuilder`] is the preferred Rust → Python → Rust embedded build -//! entry point and [`AicEngine`] is the PyO3 hot-path pyclass. +//! [`AicEngineBuilder`] is the low-level Rust → Python → Rust build path for +//! embedders that need an [`AicEngine`] step-latency handle (for example Dynamo +//! Mocker). It is not a forward-pass estimator constructor: Planner, Replay, +//! and Sweeper use [`ForwardPassPerfModel::best_available`] and the canonical +//! [`ForwardPassPerfModelConfig`] contract. //! //! This directory remains a stable mirror of the former AIConfigurator Rust //! crate. Keeping the imported implementation behind one namespace makes @@ -41,8 +44,9 @@ pub use common::AicError; // planner / Mocker) can use it natively; also exposed to Python via the // `RustForwardPassPerfModel` pyclass in `py.rs`. pub use fpm::{ - ForwardPassPerfDiagnostics, ForwardPassPerfModel, ForwardPassPerfOptions, - ForwardPassPerfReadiness, ForwardPassPerfSource, + ForwardPassFallbackPolicy, ForwardPassModelKind, ForwardPassPerfDiagnostics, + ForwardPassPerfModel, ForwardPassPerfModelConfig, ForwardPassPerfOptions, + ForwardPassPerfProvenance, ForwardPassPerfReadiness, ForwardPassPerfSource, }; // Forward-pass metrics telemetry types and schema version, plus the // crate-internal validation helper. Re-exported at the crate root so existing @@ -59,7 +63,8 @@ pub use memory::{ KvCacheEstimateOptions, KvCacheEstimateRequest, KvCacheMemoryFraction, MemoryBreakdown, }; // PyO3 bindings. `AicEngine` is the Python -> Rust hot-path pyclass; -// `AicEngineBuilder` is the Rust -> Python -> Rust entry point. They must be +// `AicEngineBuilder` is the low-level Rust -> Python -> Rust compiled-engine +// entry point. It does not construct a forward-pass estimator. They must be // `pub`-re-exported here because the `py` module itself is private. #[cfg(feature = "python")] pub use py::{AicEngine, AicEngineBuilder}; diff --git a/crates/core/src/perfmodel/py.rs b/crates/core/src/perfmodel/py.rs index d067f8c1b..9b8ba3ee0 100644 --- a/crates/core/src/perfmodel/py.rs +++ b/crates/core/src/perfmodel/py.rs @@ -13,8 +13,11 @@ //! Rust `run_agg`. Each method //! releases the GIL around the Rust compute via [`Python::allow_threads`], //! so the Rust compute runs without holding the GIL. -//! * **Rust → Python → Rust (embedded path).** [`AicEngineBuilder`] is the -//! Rust entry point. It crosses into Python once to run +//! * **Rust → Python → Rust (low-level compiled-engine path).** +//! [`AicEngineBuilder`] is reserved for embedders such as Dynamo Mocker that +//! need an [`AicEngine`] hot-path handle. It does not construct a +//! `ForwardPassPerfModel` and is not exposed to Python Planner, Replay, or +//! Sweeper. It crosses into Python once to run //! `aiconfigurator_core.sdk.engine.compile_engine`, then build an [`Engine`] //! from the returned bincode bytes. After that the `predict_*` hot path is //! pure Rust with no GIL. @@ -34,11 +37,11 @@ use pyo3::sync::GILOnceCell; use pyo3::types::PyType; use crate::common::error::AicError; -use crate::perfmodel::EngineConfig; use crate::perfmodel::engine::runtime::{ DEFAULT_STATIC_STRIDE, Engine, PerOpSolValue, PerOpValue, RuntimeConfig, StaticMode, StaticResult, }; +use crate::perfmodel::{EngineConfig, ForwardPassPerfModelConfig}; use crate::{BackendKind, DataType, ENGINE_CONFIG_SCHEMA_VERSION}; /// Trivial smoke export: returns the engine-config schema version so callers @@ -147,7 +150,7 @@ fn parse_mode(mode: &str) -> PyResult { /// Precedence: explicit `systems_path` arg → `AICONFIGURATOR_SYSTEMS_PATH` env /// → the installed core wheel's SDK resource path → repo-relative /// `python/aisimulate/src/aiconfigurator_core/systems`. -fn resolve_systems_root(systems_path: Option<&str>) -> PyResult { +pub(crate) fn resolve_systems_root(systems_path: Option<&str>) -> PyResult { if let Some(p) = systems_path { return Ok(PathBuf::from(p)); } @@ -831,13 +834,23 @@ struct EngineBuildRequest { kv_block_size: Option, systems_path: Option, forward_model: Option, + database_mode: Option, + transfer_policy: Option>, } -/// Ergonomic builder for the Rust -> Python -> Rust compiled-engine entry point. +/// Low-level builder for a Rust -> Python -> Rust [`AicEngine`] handle. /// -/// Only the model, system, and backend are required. Parallelism defaults to -/// one, speculative decoding defaults to disabled, and all other options defer -/// to Python's `compile_engine` defaults. +/// This is intentionally **not** a forward-pass estimator constructor. It is +/// retained for Rust embedders such as Dynamo Mocker that directly call the +/// compiled engine's step-latency methods. Planner, Replay, and Sweeper must use +/// [`crate::ForwardPassPerfModel::best_available`] with +/// [`crate::ForwardPassPerfModelConfig`] instead. The builder is not registered +/// on the Python extension module, which prevents those Python consumers from +/// treating its independent engine-compilation request as an estimator schema. +/// +/// Only the model, system, and backend are required for this low-level engine +/// handle. Parallelism defaults to one, speculative decoding defaults to +/// disabled, and all other options defer to Python's `compile_engine` defaults. #[derive(Clone, Debug)] pub struct AicEngineBuilder { request: EngineBuildRequest, @@ -870,6 +883,8 @@ impl AicEngineBuilder { kv_block_size: None, systems_path: None, forward_model: None, + database_mode: None, + transfer_policy: None, }, } } @@ -1047,6 +1062,8 @@ fn compile_engine_from_request(request: EngineBuildRequest) -> Result Result { + compile_engine_from_request(EngineBuildRequest { + model_path: config.model.clone(), + system: config.system.clone(), + backend: config.backend.as_str().to_owned(), + backend_version: config.backend_version.clone(), + tp_size: config.tp, + pp_size: config.pp, + attention_dp_size: config.attention_dp, + moe_tp_size: config.moe_tp_size, + moe_ep_size: config.moe_ep_size, + gemm_quant_mode: config.gemm_quant_mode.clone(), + moe_quant_mode: config.moe_quant_mode.clone(), + kvcache_quant_mode: config.kvcache_quant_mode.clone(), + fmha_quant_mode: config.fmha_quant_mode.clone(), + comm_quant_mode: config.comm_quant_mode.clone(), + nextn: config.nextn, + kv_block_size: config.kv_block_size, + systems_path: Some(systems_path.to_owned()), + forward_model: Some(config.forward_model.as_str().to_owned()), + database_mode: Some(database_mode_token(config.database_mode).to_owned()), + transfer_policy: config.transfer_policy.clone(), }) } +fn database_mode_token(mode: crate::common::enums::DatabaseMode) -> &'static str { + use crate::common::enums::DatabaseMode; + match mode { + DatabaseMode::Silicon => "SILICON", + DatabaseMode::Hybrid => "HYBRID", + DatabaseMode::Empirical => "EMPIRICAL", + DatabaseMode::Sol => "SOL", + DatabaseMode::SolFull => "SOL_FULL", + } +} + /// `DataType` → `GEMMQuantMode` enum name. `None` (auto-infer) for DataTypes /// with no GEMM equivalent. Identity for bf16/fp8/fp8_static/fp8_block/nvfp4 /// and w4a16_nvfp4; `int8`→`int8_wo`, `int4`→`int4_wo`. @@ -1178,12 +1237,12 @@ fn kvcache_quant_name(dtype: Option<&DataType>) -> Option<&'static str> { /// The hot path (`estimate_forward_pass_time_ms` / `tune_with_fpms`) is pure /// Rust over the embedded [`Engine`] with NO Python re-entry — the GIL is /// released via [`Python::allow_threads`] around each compute. Only the -/// constructors (`from_native` / `best_available`) cross into Python once to +/// sole production constructor (`best_available`) crosses into Python once to /// compile the model (`compile_engine`); that crossing re-acquires the GIL via /// `with_gil`, which is re-entrant, so calling it from inside a `#[pymethod]` /// staticmethod is fine. /// -/// Constructors take the engine config + options as JSON strings (the same +/// The constructor takes the canonical config + options as JSON strings (the same /// marshalling the Python `RustForwardPassPerfModel` wrapper used to pass over /// ctypes), so the Python wrapper's public surface is unchanged. #[pyclass(name = "RustForwardPassPerfModel")] @@ -1191,13 +1250,15 @@ pub struct PyForwardPassPerfModel { inner: crate::ForwardPassPerfModel, } -/// Parse the optional options JSON into [`ForwardPassPerfOptions`], defaulting -/// when `None`/empty. Serde fills missing fields from the per-field defaults. -fn parse_fpm_options(options_json: Option<&str>) -> PyResult { +/// Parse optional runtime/tuning options. `None` lets the core own defaults. +fn parse_fpm_options( + options_json: Option<&str>, +) -> PyResult> { match options_json { - None => Ok(crate::ForwardPassPerfOptions::default()), - Some(s) if s.trim().is_empty() => Ok(crate::ForwardPassPerfOptions::default()), + None => Ok(None), + Some(s) if s.trim().is_empty() => Ok(None), Some(s) => serde_json::from_str(s) + .map(Some) .map_err(|e| PyValueError::new_err(format!("invalid options JSON: {e}"))), } } @@ -1219,43 +1280,20 @@ fn parse_fpm_iteration(fpm_json: &str) -> PyResult) -> PyResult { - let config: EngineConfig = serde_json::from_str(config_json) - .map_err(|e| PyValueError::new_err(format!("invalid engine config JSON: {e}")))?; - let options = parse_fpm_options(options_json)?; - let inner = crate::ForwardPassPerfModel::from_native(config, options).map_err(aic_to_py)?; - Ok(Self { inner }) - } - /// `RustForwardPassPerfModel.best_available(config_json, options_json=None)`: - /// native when possible, else regression fallback (reason in - /// `diagnostics()["last_warning"]`). + /// the sole production constructor. The config's explicit fallback policy + /// decides whether unsupported native input fails or uses regression. #[staticmethod] #[pyo3(signature = (config_json, options_json=None))] fn best_available(config_json: &str, options_json: Option<&str>) -> PyResult { - let config: EngineConfig = serde_json::from_str(config_json) - .map_err(|e| PyValueError::new_err(format!("invalid engine config JSON: {e}")))?; + let config: crate::ForwardPassPerfModelConfig = serde_json::from_str(config_json) + .map_err(|e| PyValueError::new_err(format!("invalid forward-pass config JSON: {e}")))?; let options = parse_fpm_options(options_json)?; let inner = crate::ForwardPassPerfModel::best_available(config, options).map_err(aic_to_py)?; Ok(Self { inner }) } - /// `RustForwardPassPerfModel.from_regression(options_json=None)`: - /// regression-only model (no native engine, no Python compile). - #[staticmethod] - #[pyo3(signature = (options_json=None))] - fn from_regression(options_json: Option<&str>) -> PyResult { - let options = parse_fpm_options(options_json)?; - let inner = crate::ForwardPassPerfModel::from_regression(options).map_err(aic_to_py)?; - Ok(Self { inner }) - } - /// Estimate one forward-pass iteration in ms. `fpm_json` is one iteration as /// a single FPM object or a per-attention-DP-rank array. Returns `None` for /// regression models without enough data yet. Pure-Rust compute (GIL freed). diff --git a/crates/core/src/python.rs b/crates/core/src/python.rs index bfd901e19..ae223c770 100644 --- a/crates/core/src/python.rs +++ b/crates/core/src/python.rs @@ -7,6 +7,10 @@ use std::path::PathBuf; use std::sync::Arc; use crate::engine::{Backend, TimingModel, TimingModelConfig}; +use crate::perfmodel::{ + FPM_VERSION, ForwardPassMetrics, ForwardPassPerfModel, ForwardPassPerfModelConfig, + ForwardPassPerfOptions, QueuedRequestMetrics, ScheduledRequestMetrics, +}; use crate::replay::{ ReplayEngineConfig, ReplayEngineFactory, ReplayRoleConfig, ReplayRuntimeInput, ReplaySpec, ReplayTopology, Replayer, @@ -18,8 +22,8 @@ use crate::replay::{ use anyhow::{Context, Result, anyhow, ensure}; use pyo3::exceptions::PyRuntimeError; use pyo3::prelude::*; -use pyo3::types::{PyAny, PyDict, PyModule}; -use serde::Deserialize; +use pyo3::types::{PyDict, PyModule}; +use serde::{Deserialize, Serialize}; #[derive(Debug, Deserialize)] #[serde(untagged)] @@ -80,62 +84,24 @@ struct RuntimeTraffic { max_sim_time_ms: Option, } -#[derive(Debug, Clone, Deserialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] struct AicTimingConfig { - model: String, - backend: String, - system: String, - #[serde(alias = "tp_size")] - tp: u32, - #[serde(default)] - backend_version: Option, - #[serde(default = "one")] - pp: u32, - #[serde(default = "one")] - attention_dp: u32, - #[serde(default)] - moe_tp_size: Option, - #[serde(default)] - moe_ep_size: Option, - #[serde(default, alias = "gemm_quant_mode")] - gemm_dtype: Option, - #[serde(default, alias = "moe_quant_mode")] - moe_dtype: Option, - #[serde(default, alias = "fmha_quant_mode")] - fmha_dtype: Option, - #[serde(default, alias = "kvcache_quant_mode")] - kv_cache_dtype: Option, - #[serde(default, alias = "comm_quant_mode")] - comm_dtype: Option, - #[serde(default)] - nextn: u32, - #[serde(default)] - kv_block_size: Option, + #[serde(flatten)] + perf_model: ForwardPassPerfModelConfig, + #[serde(default, skip_serializing_if = "Option::is_none")] + options: Option, #[serde(default)] gpu_memory_utilization: Option, #[serde(default)] mem_fraction_static: Option, #[serde(default)] free_gpu_memory_fraction: Option, - #[serde(default)] - systems_path: Option, -} - -const fn one() -> u32 { - 1 } impl AicTimingConfig { fn resolved_backend_version(&self) -> &str { - self.backend_version - .as_deref() - .unwrap_or(match self.backend.as_str() { - "vllm" => "0.19.0", - "sglang" => "0.5.10", - "trtllm" => "1.3.0rc10", - _ => "", - }) + self.perf_model.backend_version.as_deref().unwrap_or("") } fn resolved_memory_fraction(&self) -> Result<(&'static str, f64)> { @@ -151,35 +117,37 @@ impl AicTimingConfig { "{name} must be finite and between 0 and 1" ); } - match self.backend.as_str() { + match self.perf_model.backend.as_str() { "vllm" => Ok(("of_total", self.gpu_memory_utilization.unwrap_or(0.9))), "sglang" => Ok(("of_total", self.mem_fraction_static.unwrap_or(0.88))), "trtllm" => Ok(("of_free", self.free_gpu_memory_fraction.unwrap_or(0.9))), _ => Err(anyhow!( "unsupported AIC backend {:?}; expected vllm, sglang, or trtllm", - self.backend + self.perf_model.backend )), } } fn validate_parallel_shape(&self) -> Result<()> { ensure!( - self.tp > 0 - && self.pp > 0 - && self.attention_dp > 0 - && self.moe_tp_size != Some(0) - && self.moe_ep_size != Some(0), + self.perf_model.tp > 0 + && self.perf_model.pp > 0 + && self.perf_model.attention_dp > 0 + && self.perf_model.moe_tp_size != Some(0) + && self.perf_model.moe_ep_size != Some(0), "AIC timing parallel sizes tp, pp, attention_dp, moe_tp_size, and \ moe_ep_size must be positive" ); - ensure!(self.nextn <= 5, "AIC nextn must be in 0..=5"); + ensure!(self.perf_model.nextn <= 5, "AIC nextn must be in 0..=5"); ensure!( - self.moe_tp_size.is_some() == self.moe_ep_size.is_some(), + self.perf_model.moe_tp_size.is_some() == self.perf_model.moe_ep_size.is_some(), "AIC moe_tp_size and moe_ep_size must be configured together" ); - if let (Some(moe_tp), Some(moe_ep)) = (self.moe_tp_size, self.moe_ep_size) { + if let (Some(moe_tp), Some(moe_ep)) = + (self.perf_model.moe_tp_size, self.perf_model.moe_ep_size) + { ensure!( - u64::from(self.tp) * u64::from(self.attention_dp) + u64::from(self.perf_model.tp) * u64::from(self.perf_model.attention_dp) == u64::from(moe_tp) * u64::from(moe_ep), "AIC topology requires tp * attention_dp == moe_tp_size * moe_ep_size" ); @@ -189,65 +157,30 @@ impl AicTimingConfig { } struct AicTimingModel { - engine: Py, + model: ForwardPassPerfModel, } impl AicTimingModel { - fn build(config: AicTimingConfig) -> Result { - ensure!( - !config.model.trim().is_empty(), - "AIC timing config field \"model\" cannot be empty" - ); - ensure!( - !config.system.trim().is_empty(), - "AIC timing config field \"system\" cannot be empty" - ); - config.validate_parallel_shape()?; - ensure!( - matches!(config.backend.as_str(), "vllm" | "sglang" | "trtllm"), - "unsupported AIC backend {:?}; expected vllm, sglang, or trtllm", - config.backend - ); - ensure!( - !config.resolved_backend_version().is_empty(), - "AIC backend version cannot be empty" - ); - config.resolved_memory_fraction()?; - - let engine = Python::with_gil(|py| -> PyResult> { - let sdk = PyModule::import(py, "aiconfigurator_core.sdk.engine")?; - let kwargs = PyDict::new(py); - kwargs.set_item("backend_version", config.resolved_backend_version())?; - kwargs.set_item("tp_size", config.tp)?; - kwargs.set_item("pp_size", config.pp)?; - kwargs.set_item("attention_dp_size", config.attention_dp)?; - kwargs.set_item("moe_tp_size", config.moe_tp_size)?; - kwargs.set_item("moe_ep_size", config.moe_ep_size)?; - kwargs.set_item("gemm_quant_mode", config.gemm_dtype.as_deref())?; - kwargs.set_item("moe_quant_mode", config.moe_dtype.as_deref())?; - kwargs.set_item("fmha_quant_mode", config.fmha_dtype.as_deref())?; - kwargs.set_item("kvcache_quant_mode", config.kv_cache_dtype.as_deref())?; - kwargs.set_item("comm_quant_mode", config.comm_dtype.as_deref())?; - kwargs.set_item("nextn", config.nextn)?; - kwargs.set_item("kv_block_size", config.kv_block_size)?; - kwargs.set_item("systems_path", config.systems_path.as_deref())?; - let spec = sdk.getattr("compile_engine")?.call( - ( - config.model.as_str(), - config.system.as_str(), - config.backend.as_str(), - ), - Some(&kwargs), - )?; - let aic = PyModule::import(py, "aiconfigurator_core")? - .getattr("AicEngine")? - .call_method1("from_spec", (spec, config.systems_path.as_deref()))?; - Ok(aic.unbind()) - }) - .map_err(|error| { - anyhow!("AIC timing provider could not compile the requested engine: {error}") - })?; - Ok(Self { engine }) + fn build(config: &AicTimingConfig) -> Result { + let model = + ForwardPassPerfModel::best_available(config.perf_model.clone(), config.options.clone()) + .map_err(|error| { + anyhow!( + "AIC timing provider could not construct the requested estimator: {error}" + ) + })?; + Ok(Self { model }) + } + + fn resolved_config(&self, authored: &AicTimingConfig) -> AicTimingConfig { + let mut resolved = authored.clone(); + if let Some(provenance) = self.model.provenance() { + resolved.perf_model = provenance.config.clone(); + if let Some(root) = provenance.selected_systems_root.as_ref() { + resolved.perf_model.systems_paths = vec![root.clone()]; + } + } + resolved } } @@ -261,37 +194,53 @@ impl TimingModel for AicTimingModel { let batch_size = checked_u32(batch_size, "prefill batch size")?; let mean_isl = checked_u32(mean_isl, "mean input length")?; let mean_prefix = checked_u32(mean_prefix, "mean prefix length")?; - Python::with_gil(|py| { - self.engine - .bind(py) - .call_method1( - "predict_prefill_latency", - (batch_size, mean_isl, mean_prefix), - )? - .extract::() - }) - .map_err(|error| anyhow!("AIC prefill prediction failed: {error}")) + let computed_tokens = mean_isl + .checked_sub(mean_prefix) + .context("mean prefix length exceeds mean input length")?; + let metrics = ForwardPassMetrics { + version: FPM_VERSION, + scheduled_requests: ScheduledRequestMetrics { + num_prefill_requests: batch_size, + sum_prefill_tokens: batch_size + .checked_mul(computed_tokens) + .context("prefill token sum exceeds AIC's u32 limit")?, + sum_prefill_kv_tokens: batch_size + .checked_mul(mean_prefix) + .context("prefill KV token sum exceeds AIC's u32 limit")?, + ..ScheduledRequestMetrics::default() + }, + queued_requests: QueuedRequestMetrics::default(), + ..ForwardPassMetrics::default() + }; + self.model + .estimate_forward_pass_time_ms(&[metrics]) + .map_err(|error| anyhow!("AIC prefill prediction failed: {error}"))? + .context("AIC prefill regression requires more tuning observations") } fn predict_decode_ms( &self, batch_size: usize, _active_kv_tokens: usize, - mean_context_length: usize, - _total_kv_tokens: usize, + _mean_context_length: usize, + total_kv_tokens: usize, ) -> Result { let batch_size = checked_u32(batch_size, "decode batch size")?; - let mean_context_length = checked_u32(mean_context_length, "mean context length")?; - Python::with_gil(|py| { - self.engine - .bind(py) - .call_method1( - "predict_decode_latency", - (batch_size, mean_context_length, 2), - )? - .extract::() - }) - .map_err(|error| anyhow!("AIC decode prediction failed: {error}")) + let total_kv_tokens = checked_u32(total_kv_tokens, "total decode KV tokens")?; + let metrics = ForwardPassMetrics { + version: FPM_VERSION, + scheduled_requests: ScheduledRequestMetrics { + num_decode_requests: batch_size, + sum_decode_kv_tokens: total_kv_tokens, + ..ScheduledRequestMetrics::default() + }, + queued_requests: QueuedRequestMetrics::default(), + ..ForwardPassMetrics::default() + }; + self.model + .estimate_forward_pass_time_ms(&[metrics]) + .map_err(|error| anyhow!("AIC decode prediction failed: {error}"))? + .context("AIC decode regression requires more tuning observations") } } @@ -310,26 +259,48 @@ fn estimate_aic_num_gpu_blocks(config: &AicTimingConfig, role: &ReplayRoleConfig kwargs.set_item("max_batch_size", role.rank.max_num_seqs)?; kwargs.set_item("memory_fraction_kind", memory_fraction_kind)?; kwargs.set_item("memory_fraction_value", memory_fraction_value)?; - kwargs.set_item("tp_size", config.tp)?; - kwargs.set_item("pp_size", config.pp)?; - kwargs.set_item("attention_dp_size", config.attention_dp)?; - kwargs.set_item("moe_tp_size", config.moe_tp_size)?; - kwargs.set_item("moe_ep_size", config.moe_ep_size)?; - kwargs.set_item("gemm_quant_mode", config.gemm_dtype.as_deref())?; - kwargs.set_item("moe_quant_mode", config.moe_dtype.as_deref())?; - kwargs.set_item("fmha_quant_mode", config.fmha_dtype.as_deref())?; - kwargs.set_item("kvcache_quant_mode", config.kv_cache_dtype.as_deref())?; - kwargs.set_item("comm_quant_mode", config.comm_dtype.as_deref())?; + kwargs.set_item("tp_size", config.perf_model.tp)?; + kwargs.set_item("pp_size", config.perf_model.pp)?; + kwargs.set_item("attention_dp_size", config.perf_model.attention_dp)?; + kwargs.set_item("moe_tp_size", config.perf_model.moe_tp_size)?; + kwargs.set_item("moe_ep_size", config.perf_model.moe_ep_size)?; + kwargs.set_item( + "gemm_quant_mode", + config.perf_model.gemm_quant_mode.as_deref(), + )?; + kwargs.set_item( + "moe_quant_mode", + config.perf_model.moe_quant_mode.as_deref(), + )?; + kwargs.set_item( + "fmha_quant_mode", + config.perf_model.fmha_quant_mode.as_deref(), + )?; + kwargs.set_item( + "kvcache_quant_mode", + config.perf_model.kvcache_quant_mode.as_deref(), + )?; + kwargs.set_item( + "comm_quant_mode", + config.perf_model.comm_quant_mode.as_deref(), + )?; // Capacity intentionally omits NextN until AIC's Eagle memory model no // longer returns negative KV capacity. Timing compilation still uses it. - kwargs.set_item("systems_path", config.systems_path.as_deref())?; + kwargs.set_item( + "systems_path", + config + .perf_model + .systems_paths + .first() + .and_then(|path| path.to_str()), + )?; memory .getattr("estimate_num_gpu_blocks")? .call( ( - config.model.as_str(), - config.system.as_str(), - config.backend.as_str(), + config.perf_model.model.as_str(), + config.perf_model.system.as_str(), + config.perf_model.backend.as_str(), ), Some(&kwargs), )? @@ -351,24 +322,25 @@ fn materialize_aic_capacity( Backend::Trtllm => "trtllm", }; ensure!( - config.backend == engine_backend, + config.perf_model.backend.as_str() == engine_backend, "AIC backend {:?} does not match engine backend {engine_backend:?}", - config.backend + config.perf_model.backend ); ensure!( - config.tp == role.tensor_parallel_size, + config.perf_model.tp == role.tensor_parallel_size, "AIC tp={} does not match engine tensor_parallel_size={}", - config.tp, + config.perf_model.tp, role.tensor_parallel_size ); ensure!( - config.attention_dp == role.dp_size, + config.perf_model.attention_dp == role.dp_size, "AIC attention_dp={} does not match engine dp_size={}", - config.attention_dp, + config.perf_model.attention_dp, role.dp_size ); ensure!( config + .perf_model .kv_block_size .is_none_or(|block_size| block_size as usize == role.rank.block_size), "AIC kv_block_size does not match engine block_size={}", @@ -376,9 +348,9 @@ fn materialize_aic_capacity( ); let engine_nextn = role.rank.aic_nextn.unwrap_or(0); ensure!( - config.nextn as usize == engine_nextn, + config.perf_model.nextn as usize == engine_nextn, "AIC nextn={} does not match engine aic_nextn={engine_nextn}", - config.nextn + config.perf_model.nextn ); if capacity_is_explicit { return Ok(()); @@ -419,15 +391,22 @@ fn resolve_role_timing( "native timing provider {provider:?} is not installed; only \"aic\" is \ available in the AISimulate runtime" ); - let config: AicTimingConfig = + let authored: AicTimingConfig = serde_json::from_value(config).context("invalid AIC timing provider configuration")?; + let timing = AicTimingModel::build(&authored)?; + let resolved = timing.resolved_config(&authored); materialize_aic_capacity( - &config, + &resolved, role, capacity_is_explicit, estimate_aic_num_gpu_blocks, )?; - Ok(Some(Arc::new(AicTimingModel::build(config)?))) + role.rank.timing_model = TimingModelConfig::External { + provider, + config: serde_json::to_value(&resolved) + .context("serializing resolved AIC timing provider configuration")?, + }; + Ok(Some(Arc::new(timing))) } fn runtime_paths(traffic: &RuntimeTraffic) -> Result> { @@ -857,29 +836,58 @@ mod tests { fn aic_config() -> AicTimingConfig { AicTimingConfig { - model: "test-model".into(), - backend: "vllm".into(), - system: "test-system".into(), - tp: 1, - backend_version: None, - pp: 1, - attention_dp: 1, - moe_tp_size: None, - moe_ep_size: None, - gemm_dtype: None, - moe_dtype: None, - fmha_dtype: None, - kv_cache_dtype: None, - comm_dtype: None, - nextn: 0, - kv_block_size: None, + perf_model: serde_json::from_value(serde_json::json!({ + "model": "test-model", + "backend": "vllm", + "system": "test-system", + "backend_version": "test-version" + })) + .unwrap(), + options: None, gpu_memory_utilization: None, mem_fraction_static: None, free_gpu_memory_fraction: None, - systems_path: None, } } + #[test] + fn aic_timing_config_round_trips_the_canonical_contract() { + let authored = serde_json::json!({ + "model": "test-model", + "backend": "vllm", + "system": "test-system", + "backend_version": "test-version", + "tp": 4, + "pp": 1, + "attention_dp": 2, + "moe_tp_size": 2, + "moe_ep_size": 4, + "forward_model": "op_level", + "database_mode": "HYBRID", + "transfer_policy": ["xshape", "xquant"], + "systems_paths": ["/tmp/aic-systems"], + "fallback_policy": "error", + "options": {"min_observations": 7}, + "gpu_memory_utilization": 0.85 + }); + let parsed: AicTimingConfig = serde_json::from_value(authored.clone()).unwrap(); + assert_eq!(parsed.perf_model.tp, 4); + assert_eq!(parsed.perf_model.attention_dp, 2); + assert_eq!(parsed.options.as_ref().unwrap().min_observations, 7); + assert_eq!(parsed.gpu_memory_utilization, Some(0.85)); + + let round_trip = serde_json::to_value(parsed).unwrap(); + assert_eq!(round_trip["model"], authored["model"]); + assert_eq!(round_trip["database_mode"], authored["database_mode"]); + assert_eq!(round_trip["options"]["min_observations"], 7); + let reparsed: AicTimingConfig = serde_json::from_value(round_trip).unwrap(); + assert_eq!(reparsed.perf_model.tp, 4); + assert_eq!( + reparsed.perf_model.transfer_policy.as_deref(), + Some(["xshape".into(), "xquant".into()].as_slice()) + ); + } + #[test] fn capacity_is_estimated_only_when_not_explicit() { let mut role = aggregated_role(&ReplayEngineConfig::default()); diff --git a/crates/tests/public-api/src/lib.rs b/crates/tests/public-api/src/lib.rs index 3f607eedc..1b84626ee 100644 --- a/crates/tests/public-api/src/lib.rs +++ b/crates/tests/public-api/src/lib.rs @@ -4,7 +4,8 @@ //! Compile-time contract tests from an external crate's point of view. use aiconfigurator_core::{ - AicEngine, AicEngineBuilder, AicError, BackendKind, ForwardPassPerfModel, + AicEngine, AicEngineBuilder, AicError, BackendKind, ForwardPassFallbackPolicy, + ForwardPassModelKind, ForwardPassPerfModel, ForwardPassPerfModelConfig, ForwardPassPerfOptions, KvCacheEstimateRequest, }; @@ -33,9 +34,43 @@ pub fn build_engine(builder: AicEngineBuilder) -> Result { builder.build() } -/// Compile the forward-pass model's public constructor and telemetry type. -pub fn regression_model() -> Result { - ForwardPassPerfModel::from_regression(ForwardPassPerfOptions::default()) +/// Compile the forward-pass model's sole public production constructor. +/// +/// The function is intentionally not called because construction loads model +/// and systems data. Its signature proves an external crate can use the +/// canonical config/options boundary without reaching internal constructors. +pub fn forward_pass_model( + config: ForwardPassPerfModelConfig, + options: Option, +) -> Result { + ForwardPassPerfModel::best_available(config, options) +} + +/// Compile construction of the canonical config from an external crate. +pub fn forward_pass_config() -> ForwardPassPerfModelConfig { + ForwardPassPerfModelConfig { + model: "Qwen/Qwen3-32B".into(), + system: "h200_sxm".into(), + backend: BackendKind::Vllm, + backend_version: Some("0.10.2".into()), + tp: 2, + pp: 1, + attention_dp: 1, + moe_tp_size: None, + moe_ep_size: None, + gemm_quant_mode: None, + moe_quant_mode: None, + fmha_quant_mode: None, + kvcache_quant_mode: None, + comm_quant_mode: None, + nextn: 0, + kv_block_size: Some(16), + forward_model: ForwardPassModelKind::OpLevel, + database_mode: Default::default(), + transfer_policy: None, + systems_paths: Vec::new(), + fallback_policy: ForwardPassFallbackPolicy::Error, + } } /// Keep the KV request type in the external-consumer contract without @@ -82,7 +117,9 @@ mod tests { } #[test] - fn regression_constructor_is_environment_independent() { - let _model = regression_model().expect("construct regression model"); + fn canonical_forward_pass_config_is_external() { + let config = forward_pass_config(); + assert_eq!(config.tp, 2); + let _constructor = forward_pass_model; } } diff --git a/docs/cli/migrate-from-aiconfigurator.md b/docs/cli/migrate-from-aiconfigurator.md index 9b7648ee0..3061142da 100644 --- a/docs/cli/migrate-from-aiconfigurator.md +++ b/docs/cli/migrate-from-aiconfigurator.md @@ -20,6 +20,7 @@ by the unified command surface. A legacy command becomes an AISimulate recommend | `--model-path` | `engine.model` | Same model identifier | | `--system` | `engine.hardware` | Same hardware identifier | | `--backend` | `engine.backend` | One backend or an explicit recommendation domain | +| `--backend-version` | `engine.backend_version` | Pin the selected backend's forward-pass performance-data version | | `--total-gpus` | `optimization.constraints.max_candidate_gpus` | Maximum GPUs per candidate | | `--isl` | `traffic.source.input_tokens` | Synthetic input length | | `--osl` | `traffic.source.output_tokens` | Synthetic output length | @@ -28,6 +29,12 @@ by the unified command surface. A legacy command becomes an AISimulate recommend | `--request-latency` | `evaluation.sla.e2e_ms` plus `optimization.strict_sla: true` | Fixed-output synthetic migration; applies per-request E2E and filters aggregate mean E2E | | `--strict-sla` | `optimization.strict_sla: true` | Reject before scalar ranking or Pareto dominance | +AISimulate uses only `engine.backend_version` for this identity. The current performance database +is keyed by backend and version, so there is no separate `performance_data_version` field. When +`engine.backend_version` is omitted, the Sweeper resolves the latest available version once before +the search begins and records that concrete version on every candidate. Pin `engine.backend` to one +concrete backend when setting `engine.backend_version`. + ## Illustrative strict SLA translation The examples below demonstrate how the strict-SLA fields map, but they are not behaviorally diff --git a/docs/sweeper/architecture.md b/docs/sweeper/architecture.md index 18d5dd99c..41767e44c 100644 --- a/docs/sweeper/architecture.md +++ b/docs/sweeper/architecture.md @@ -30,7 +30,8 @@ flowchart TD C --> D["Resolve configured providers"] D --> E["Generate namespaced search dimensions"] E --> F["Ask sampler for suggestions"] - F --> G["Materialize backend and adapter config"] + F --> R["Resolve exact per-role estimator identities through Core"] + R --> G["Materialize backend and adapter config"] G --> H["Build ReplaySpec"] H --> I["Worker-local Runner executes replay"] I --> J["Score and tell sampler"] @@ -42,6 +43,24 @@ Provider code runs in the main process. Worker tasks receive only a serializable do not import or pickle provider objects. Each worker creates one runner and reuses it for candidate replays. +AIConfigurator Core owns the typed `ForwardPassPerfModelConfig` (immutable identity and selection +policy), `ForwardPassPerfOptions` (tuning behavior), and the sole production constructor, +`ForwardPassPerfModel::best_available`. Sweeper parses YAML into those types and calls the Core +constructor only after a suggestion has concrete aggregated/prefill/decode topology and block-size +values. Resolution happens on the main process before the candidate reaches a replay runner, and +identical exact requests are cached for the rest of the run. + +`ReplaySpec.backend_deployment.forward_pass_estimators` stores one Core-resolved config, options, +and diagnostics/provenance record for every engine role. Sweeper does not edit those returned +configs. The same exact role config is used both as Replay's AIC timing-provider config and as +performance-model metadata, so worker processes never consult mutable global system paths or +choose a newer data version independently. Planner and other estimator consumers use the same +public Core facade instead of defining a parallel constructor schema. + +`AicEngineBuilder` is separate and deliberately narrower: it is a Rust-only, low-level compiled +`AicEngine` handle for Mocker-style step-latency embedders. It is not registered on the Python +extension and is not an estimator construction path for Planner, Replay, or Sweeper. + ## Provider Preparation A provider implements two operations: diff --git a/docs/sweeper/configuration.md b/docs/sweeper/configuration.md index 7480c4273..ac026b2e2 100644 --- a/docs/sweeper/configuration.md +++ b/docs/sweeper/configuration.md @@ -20,6 +20,16 @@ search_space: gpu_budget: 32 deployment_mode: [disagg, agg] backend: [vllm, sglang] + backend_version: + vllm: 0.11.0 + sglang: 0.5.6 + database_mode: HYBRID + transfer_policy: balanced + forward_model: op_level + forward_pass_fallback_policy: error + forward_pass_options: + min_observations: 5 + systems_paths: [default] adapters: example.policy: @@ -59,6 +69,13 @@ configuration for each candidate. | `hardware_sku` | required | AI Configurator system identifier | | `deployment_mode` | `[disagg, agg]` | deployment branches to search | | `backend` | `[vllm]` | engine backends to search | +| `backend_version` | `None` | exact version for one backend, or a per-backend version mapping; omitted backends resolve once to latest | +| `database_mode` | `SILICON` | forward-pass estimator data-source policy; see below | +| `transfer_policy` | `None` (all) | Core-owned empirical-transfer preset or tier list; used only by `HYBRID` and `EMPIRICAL` | +| `forward_model` | `op_level` | granular `op_level` or exact-data `fpm` forward estimation | +| `forward_pass_fallback_policy` | `error` | fail closed, or explicitly use observation-gated `regression` when native construction is unsupported | +| `forward_pass_options` | `None` (Core defaults) | runtime tuning controls such as observation limits, regression buckets, correction bounds, and workload-axis capacity | +| `systems_paths` | `[default]` | ordered request-scoped system/data roots; `default` is the packaged Core root | | `gpu_budget` | `32` | maximum GPUs per candidate | | `min_gpu_budget` | `None` | optional lower bound during enumeration | | `context_length` | `None` | optional KV-feasibility sequence length | @@ -69,6 +86,30 @@ configuration for each candidate. Each engine role also has lists for `max_num_batched_tokens` and `max_num_seqs`, plus pinned block size, GPU-memory-utilization, and prefix-caching fields. A one-item list pins a searched field. +Estimator controls are parsed once, then each concrete candidate role resolves through Core after +its TP, PP, attention-DP, MoE parallelism, and block size are known. `latest` becomes the exact +backend/performance-data version returned by Core, custom system paths remain request-scoped, and +every `ReplaySpec` plus returned candidate records the same unmodified resolved config, options, +and provenance. An unavailable exact identity fails candidate materialization before the replay +runner executes that trial. Regression is never an implicit degradation: it must be requested with +`forward_pass_fallback_policy: regression`, and it remains unready until `tune_with_fpms` supplies +enough observations for the workload kind. + +Database modes choose the source of each operation estimate: + +| Mode | Resolution | +|---|---| +| `SILICON` | collected performance data and supported interpolation only | +| `HYBRID` | collected data first; calibrated empirical estimation for uncovered operations | +| `EMPIRICAL` | calibrated estimation for every operation (`latency = SOL / utilization`) | +| `SOL` | uncalibrated analytic speed-of-light estimate | + +`transfer_policy` is not a search dimension. It selects which fixed Core transfer kinds the +empirical forward-pass estimator may use when its own calibration slice is missing. It accepts `off`, +`conservative`, `balanced`, or `aggressive`, or an explicit list containing `xshape`, `xquant`, +`xprofile`, and `xop`. Core validates the request and the Sweeper records the normalized explicit +policy in every candidate. The field is ignored by `SILICON` and `SOL`. + ## Pinned Parallel Configurations Pinning `parallel_configs` requires exactly one deployment mode. An aggregated entry is one shape: diff --git a/python/aisimulate/src/aiconfigurator_core/_aiconfigurator_core.pyi b/python/aisimulate/src/aiconfigurator_core/_aiconfigurator_core.pyi index d55b65ca6..13b8e714c 100644 --- a/python/aisimulate/src/aiconfigurator_core/_aiconfigurator_core.pyi +++ b/python/aisimulate/src/aiconfigurator_core/_aiconfigurator_core.pyi @@ -114,12 +114,8 @@ class AicEngine: def last_provenance(self) -> str | None: ... class RustForwardPassPerfModel: - @staticmethod - def from_native(config_json: str, options_json: str | None = None) -> RustForwardPassPerfModel: ... @staticmethod def best_available(config_json: str, options_json: str | None = None) -> RustForwardPassPerfModel: ... - @staticmethod - def from_regression(options_json: str | None = None) -> RustForwardPassPerfModel: ... def estimate_forward_pass_time_ms(self, fpm_json: str) -> float | None: ... def tune_with_fpms(self, iterations_json: str) -> None: ... def diagnostics(self) -> str: ... diff --git a/python/aisimulate/src/aiconfigurator_core/sdk/__init__.py b/python/aisimulate/src/aiconfigurator_core/sdk/__init__.py index b23ffc10b..651353116 100644 --- a/python/aisimulate/src/aiconfigurator_core/sdk/__init__.py +++ b/python/aisimulate/src/aiconfigurator_core/sdk/__init__.py @@ -20,6 +20,8 @@ __all__ = [ "EngineHandle", + "ForwardPassPerfModelConfig", + "ForwardPassPerfOptions", "ModelConfig", "RuntimeConfig", "RustForwardPassPerfModel", @@ -30,6 +32,14 @@ _PUBLIC_EXPORTS = { "EngineHandle": ("aiconfigurator_core.sdk.engine", "EngineHandle"), + "ForwardPassPerfModelConfig": ( + "aiconfigurator_core.sdk.rust_engine_step", + "ForwardPassPerfModelConfig", + ), + "ForwardPassPerfOptions": ( + "aiconfigurator_core.sdk.rust_engine_step", + "ForwardPassPerfOptions", + ), "ModelConfig": ("aiconfigurator_core.sdk.config", "ModelConfig"), "RuntimeConfig": ("aiconfigurator_core.sdk.config", "RuntimeConfig"), "RustForwardPassPerfModel": ( @@ -64,4 +74,8 @@ def __dir__() -> list[str]: from aiconfigurator_core.sdk.config import ModelConfig, RuntimeConfig from aiconfigurator_core.sdk.engine import EngineHandle, compile_engine from aiconfigurator_core.sdk.memory import estimate_kv_cache, estimate_num_gpu_blocks - from aiconfigurator_core.sdk.rust_engine_step import RustForwardPassPerfModel + from aiconfigurator_core.sdk.rust_engine_step import ( + ForwardPassPerfModelConfig, + ForwardPassPerfOptions, + RustForwardPassPerfModel, + ) diff --git a/python/aisimulate/src/aiconfigurator_core/sdk/engine.py b/python/aisimulate/src/aiconfigurator_core/sdk/engine.py index 4343fd234..825a77858 100644 --- a/python/aisimulate/src/aiconfigurator_core/sdk/engine.py +++ b/python/aisimulate/src/aiconfigurator_core/sdk/engine.py @@ -320,6 +320,8 @@ def compile_engine( kv_block_size: int | None = None, systems_path: str | None = None, forward_model: str | None = None, + database_mode: str | None = None, + transfer_policy: list[str] | None = None, ) -> bytes: """Compile a model into bincoded ``EngineSpec`` bytes. @@ -352,18 +354,28 @@ def compile_engine( model = get_model(model_path, model_config, backend) # The database supplies the shared-layer perf sources, the query mode and - # the transfer policy stamped into the compiled `EngineConfig`. Load lazily - # and tolerate failure; the Rust core falls back to its own defaults. - database = _maybe_load_database(system, backend, backend_version, systems_path) + # the transfer policy stamped into the compiled `EngineConfig`. Explicit + # construction policy is fail-closed; legacy callers without one retain + # the historical best-effort behavior. + database = _maybe_load_database( + system, + backend, + backend_version, + systems_path, + database_mode=database_mode, + transfer_policy=transfer_policy, + ) + resolved_backend_version = getattr(database, "version", None) or backend_version + resolved_systems_path = getattr(database, "systems_root", None) or systems_path spec_json = build_engine_spec_json( model, model_path=model_path, system=system, backend=backend, - backend_version=backend_version, + backend_version=resolved_backend_version, kv_block_size=kv_block_size, - systems_path=systems_path, + systems_path=resolved_systems_path, nextn=model_config.nextn, database=database, ) @@ -605,12 +617,40 @@ def _evaluate_single_op( return PerformanceResult(latency, energy=energy, source=source) -def _maybe_load_database(system: str, backend: str, backend_version: str | None, systems_path: str | None) -> Any: +def _maybe_load_database( + system: str, + backend: str, + backend_version: str | None, + systems_path: str | None, + *, + database_mode: str | None = None, + transfer_policy: list[str] | None = None, +) -> Any: try: from aiconfigurator_core.sdk import perf_database - return perf_database.get_database(system, backend, backend_version, systems_paths=systems_path) + resolved_version = backend_version or perf_database.get_latest_database_version( + system, + backend, + systems_paths=systems_path, + ) + if resolved_version is None: + return None + if database_mode is None and transfer_policy is None: + return perf_database.get_database(system, backend, resolved_version, systems_paths=systems_path) + mode = (database_mode or "SILICON").upper() + return perf_database.get_database_view( + system, + backend, + resolved_version, + systems_paths=systems_path, + allow_missing_data=mode != "SILICON", + database_mode=mode, + transfer_policy=transfer_policy, + ) except Exception: + if database_mode is not None or transfer_policy is not None: + raise return None diff --git a/python/aisimulate/src/aiconfigurator_core/sdk/rust_engine_step.py b/python/aisimulate/src/aiconfigurator_core/sdk/rust_engine_step.py index 66f62c7c9..f27530a6e 100644 --- a/python/aisimulate/src/aiconfigurator_core/sdk/rust_engine_step.py +++ b/python/aisimulate/src/aiconfigurator_core/sdk/rust_engine_step.py @@ -18,6 +18,8 @@ import os import threading from collections import OrderedDict +from collections.abc import Mapping +from dataclasses import asdict, dataclass from importlib import resources as pkg_resources from pathlib import Path from typing import Any @@ -82,6 +84,56 @@ class RustEngineUnsupportedError(RuntimeError): error-symmetric between the engines.""" +@dataclass(frozen=True) +class ForwardPassPerfModelConfig: + """Canonical immutable identity and selection policy for the estimator.""" + + model: str + system: str + backend: str + backend_version: str | None = None + tp: int = 1 + pp: int = 1 + attention_dp: int = 1 + moe_tp_size: int | None = None + moe_ep_size: int | None = None + gemm_quant_mode: str | None = None + moe_quant_mode: str | None = None + fmha_quant_mode: str | None = None + kvcache_quant_mode: str | None = None + comm_quant_mode: str | None = None + nextn: int = 0 + kv_block_size: int | None = None + forward_model: str = "op_level" + database_mode: str = "SILICON" + transfer_policy: str | tuple[str, ...] | None = None + systems_paths: tuple[str, ...] = () + fallback_policy: str = "error" + + def to_dict(self) -> dict[str, Any]: + payload = asdict(self) + payload["transfer_policy"] = _resolve_forward_pass_transfer_policy(self.transfer_policy) + payload["systems_paths"] = _resolve_forward_pass_systems_paths(self.systems_paths) + return payload + + +@dataclass(frozen=True) +class ForwardPassPerfOptions: + """Runtime observation, regression, correction, and capacity controls.""" + + max_observations: int = 64 + min_observations: int = 5 + min_faster_correction_factor: float | None = 0.5 + max_slower_correction_factor: float | None = 2.0 + bucket_count: int = 16 + max_num_tokens: int = 8192 + max_batch_size: int = 512 + max_kv_tokens: int = 2_000_000 + + def to_dict(self) -> dict[str, Any]: + return asdict(self) + + class RustForwardPassPerfModel: """Facade over the compiled Rust forward-pass perf model (PR #1152). @@ -134,69 +186,30 @@ class RustForwardPassPerfModel: def __init__(self, inner: Any) -> None: self._inner = inner - @classmethod - def from_native( - cls, - config: dict[str, Any], - options: dict[str, Any] | None = None, - ) -> RustForwardPassPerfModel: - """API: ``RustForwardPassPerfModel.from_native(config, options=None)``. - - Description: create a strict native AIC forward-pass model. - - Crosses into the Rust core, which compiles ``config`` via - ``aiconfigurator_core.sdk.engine.compile_engine``. Raises if the config is - unsupported by the native estimator. Use ``best_available()`` when - unsupported configs should fall back to the learned regression model. - """ - _configure_default_data_roots() - import aiconfigurator_core - - inner = aiconfigurator_core.RustForwardPassPerfModel.from_native( - _json_dumps(config), - _optional_json_dumps(options), - ) - return cls(inner) - @classmethod def best_available( cls, - config: dict[str, Any], - options: dict[str, Any] | None = None, + config: ForwardPassPerfModelConfig | Mapping[str, Any], + options: ForwardPassPerfOptions | Mapping[str, Any] | None = None, ) -> RustForwardPassPerfModel: """API: ``RustForwardPassPerfModel.best_available(config, options=None)``. - Description: create a native model when possible, otherwise fall back to - regression. Fallback reason is available from - ``diagnostics()["last_warning"]``. + This is the only production constructor. ``config`` owns immutable + identity and selection policy; ``options`` owns runtime tuning controls. + Regression fallback occurs only when ``fallback_policy="regression"``. """ _configure_default_data_roots() import aiconfigurator_core - inner = aiconfigurator_core.RustForwardPassPerfModel.best_available( - _json_dumps(config), - _optional_json_dumps(options), + config_payload = config.to_dict() if isinstance(config, ForwardPassPerfModelConfig) else dict(config) + config_payload["systems_paths"] = _resolve_forward_pass_systems_paths( + tuple(config_payload.get("systems_paths") or ()) ) - return cls(inner) - - @classmethod - def from_regression( - cls, - options: dict[str, Any] | None = None, - ) -> RustForwardPassPerfModel: - """API: ``RustForwardPassPerfModel.from_regression(options=None)``. - - Description: create a regression-only forward-pass model. Regression - models return ``None`` for non-empty estimates until enough samples have - been provided for the inferred workload kind through - ``tune_with_fpms()``. Correction factor getters return ``None`` in this - mode. - """ - _configure_default_data_roots() - import aiconfigurator_core - - inner = aiconfigurator_core.RustForwardPassPerfModel.from_regression( - _optional_json_dumps(options), + config_payload["transfer_policy"] = _resolve_forward_pass_transfer_policy(config_payload.get("transfer_policy")) + options_payload = options.to_dict() if isinstance(options, ForwardPassPerfOptions) else options + inner = aiconfigurator_core.RustForwardPassPerfModel.best_available( + _json_dumps(config_payload), + _optional_json_dumps(options_payload), ) return cls(inner) @@ -268,12 +281,34 @@ def _json_dumps(value: Any) -> str: return json.dumps(value, separators=(",", ":"), sort_keys=True) -def _optional_json_dumps(value: dict[str, Any] | None) -> str | None: +def _optional_json_dumps(value: Mapping[str, Any] | None) -> str | None: if value is None: return None return _json_dumps(value) +def _resolve_forward_pass_systems_paths(entries: tuple[str, ...]) -> list[str]: + packaged = os.fspath(pkg_resources.files("aiconfigurator_core") / "systems") + resolved: list[str] = [] + for entry in entries or (packaged,): + path = packaged if entry.lower() == "default" else os.path.abspath(os.path.expanduser(entry)) + if not os.path.isdir(path): + raise ValueError(f"forward-pass systems path is not a directory: {path}") + if path not in resolved: + resolved.append(path) + return resolved + + +def _resolve_forward_pass_transfer_policy(value: Any) -> list[str] | None: + if value is None: + return None + if isinstance(value, str): + from aiconfigurator_core.sdk.common import resolve_transfer_policy + + return list(resolve_transfer_policy(value)) + return [str(token) for token in value] + + def _normalize_tuning_iterations(iterations: dict[str, Any] | list[Any]) -> list[Any]: if isinstance(iterations, dict): return [[iterations]] diff --git a/python/aisimulate/src/aisimulate/aic.py b/python/aisimulate/src/aisimulate/aic.py index 9b1704790..3f9890be0 100644 --- a/python/aisimulate/src/aisimulate/aic.py +++ b/python/aisimulate/src/aisimulate/aic.py @@ -31,10 +31,19 @@ def materialize_aic_num_gpu_blocks(raw: dict[str, Any]) -> dict[str, Any]: """Return engine arguments with rank-local AIC KV capacity materialized.""" lowered = dict(raw) - attention_dp = lowered.get("aic_attention_dp_size") + timing_model = lowered.get("timing_model") + timing_config = ( + timing_model.get("config") + if isinstance(timing_model, dict) + and timing_model.get("type") == "external" + and timing_model.get("provider") == "aic" + and isinstance(timing_model.get("config"), dict) + else {} + ) + attention_dp = lowered.get("aic_attention_dp_size", timing_config.get("attention_dp")) dp = attention_dp or 1 configured_dp = lowered.get("dp_size") or 1 - has_aic_config = lowered.get("aic_backend") is not None or attention_dp is not None + has_aic_config = lowered.get("aic_backend") is not None or bool(timing_config) if has_aic_config and configured_dp > 1 and configured_dp != dp: raise ValueError( "dp_size must match aic_attention_dp_size for AIC-backed replay " @@ -45,28 +54,23 @@ def materialize_aic_num_gpu_blocks(raw: dict[str, Any]) -> dict[str, Any]: if lowered.get("num_gpu_blocks") is not None: return lowered - backend = lowered.get("aic_backend") + backend = lowered.get("aic_backend", timing_config.get("backend")) if backend is None: return lowered if not isinstance(backend, str) or backend not in DEFAULT_BACKEND_VERSIONS: supported = ", ".join(sorted(DEFAULT_BACKEND_VERSIONS)) raise ValueError( - f"AIC KV cache capacity estimation does not support {backend!r}; " - f"supported backends: {supported}" + f"AIC KV cache capacity estimation does not support {backend!r}; supported backends: {supported}" ) - model = lowered.get("aic_model_path") + model = lowered.get("aic_model_path", timing_config.get("model")) if not model: - raise ValueError( - "AIC KV cache capacity estimation requires aic_model_path in engine args" - ) + raise ValueError("AIC KV cache capacity estimation requires aic_model_path in engine args") lowered["num_gpu_blocks"] = estimate_num_gpu_blocks( backend_name=backend, - system=lowered.get("aic_system") or _DEFAULT_AIC_SYSTEM, + system=lowered.get("aic_system", timing_config.get("system")) or _DEFAULT_AIC_SYSTEM, model_path=model, - tp_size=( - lowered.get("aic_tp_size") if lowered.get("aic_tp_size") is not None else 1 - ), + tp_size=lowered.get("aic_tp_size", timing_config.get("tp", 1)), block_size=_resolve_block_size(lowered, backend), max_num_batched_tokens=( lowered.get("max_num_batched_tokens") @@ -74,26 +78,22 @@ def materialize_aic_num_gpu_blocks(raw: dict[str, Any]) -> dict[str, Any]: else _DEFAULT_MAX_NUM_BATCHED_TOKENS ), max_num_sequences=( - lowered.get("max_num_seqs") - if lowered.get("max_num_seqs") is not None - else _DEFAULT_MAX_NUM_SEQUENCES + lowered.get("max_num_seqs") if lowered.get("max_num_seqs") is not None else _DEFAULT_MAX_NUM_SEQUENCES ), - gpu_memory_utilization=lowered.get("gpu_memory_utilization"), - mem_fraction_static=lowered.get("mem_fraction_static"), - free_gpu_memory_fraction=lowered.get("free_gpu_memory_fraction"), - backend_version=lowered.get("aic_backend_version"), - pp_size=( - lowered.get("aic_pp_size") if lowered.get("aic_pp_size") is not None else 1 - ), - moe_tp_size=lowered.get("aic_moe_tp_size"), - moe_ep_size=lowered.get("aic_moe_ep_size"), + gpu_memory_utilization=lowered.get("gpu_memory_utilization", timing_config.get("gpu_memory_utilization")), + mem_fraction_static=lowered.get("mem_fraction_static", timing_config.get("mem_fraction_static")), + free_gpu_memory_fraction=lowered.get("free_gpu_memory_fraction", timing_config.get("free_gpu_memory_fraction")), + backend_version=lowered.get("aic_backend_version", timing_config.get("backend_version")), + pp_size=lowered.get("aic_pp_size", timing_config.get("pp", 1)), + moe_tp_size=lowered.get("aic_moe_tp_size", timing_config.get("moe_tp_size")), + moe_ep_size=lowered.get("aic_moe_ep_size", timing_config.get("moe_ep_size")), attention_dp_size=attention_dp, - gemm_dtype=lowered.get("aic_gemm_dtype"), - moe_dtype=lowered.get("aic_moe_dtype"), - fmha_dtype=lowered.get("aic_fmha_dtype"), - kv_cache_dtype=lowered.get("aic_kv_cache_dtype"), - comm_dtype=lowered.get("aic_comm_dtype"), - systems_path=lowered.get("systems_path"), + gemm_dtype=lowered.get("aic_gemm_dtype", timing_config.get("gemm_quant_mode")), + moe_dtype=lowered.get("aic_moe_dtype", timing_config.get("moe_quant_mode")), + fmha_dtype=lowered.get("aic_fmha_dtype", timing_config.get("fmha_quant_mode")), + kv_cache_dtype=lowered.get("aic_kv_cache_dtype", timing_config.get("kvcache_quant_mode")), + comm_dtype=lowered.get("aic_comm_dtype", timing_config.get("comm_quant_mode")), + systems_path=(lowered.get("systems_path") or next(iter(timing_config.get("systems_paths") or ()), None)), ) return lowered @@ -132,8 +132,7 @@ def estimate_num_gpu_blocks( if backend_name not in DEFAULT_BACKEND_VERSIONS: supported = ", ".join(sorted(DEFAULT_BACKEND_VERSIONS)) raise ValueError( - f"AIC KV cache capacity estimation does not support {backend_name!r}; " - f"supported backends: {supported}" + f"AIC KV cache capacity estimation does not support {backend_name!r}; supported backends: {supported}" ) from aiconfigurator_core.sdk.memory import ( estimate_num_gpu_blocks as aic_estimate_num_gpu_blocks, @@ -142,23 +141,15 @@ def estimate_num_gpu_blocks( if backend_name == "trtllm": memory_fraction_kind = "of_free" memory_fraction_value = ( - free_gpu_memory_fraction - if free_gpu_memory_fraction is not None - else DEFAULT_FREE_GPU_MEMORY_FRACTION + free_gpu_memory_fraction if free_gpu_memory_fraction is not None else DEFAULT_FREE_GPU_MEMORY_FRACTION ) elif backend_name == "sglang": memory_fraction_kind = "of_total" - memory_fraction_value = ( - mem_fraction_static - if mem_fraction_static is not None - else DEFAULT_MEM_FRACTION_STATIC - ) + memory_fraction_value = mem_fraction_static if mem_fraction_static is not None else DEFAULT_MEM_FRACTION_STATIC else: memory_fraction_kind = "of_total" memory_fraction_value = ( - gpu_memory_utilization - if gpu_memory_utilization is not None - else DEFAULT_GPU_MEMORY_UTILIZATION + gpu_memory_utilization if gpu_memory_utilization is not None else DEFAULT_GPU_MEMORY_UTILIZATION ) return int( @@ -167,9 +158,7 @@ def estimate_num_gpu_blocks( system, backend_name, backend_version=( - backend_version - if backend_version is not None - else DEFAULT_BACKEND_VERSIONS[backend_name] + backend_version if backend_version is not None else DEFAULT_BACKEND_VERSIONS[backend_name] ), scheduler_block_size=block_size, max_num_tokens=max_num_batched_tokens, @@ -178,9 +167,7 @@ def estimate_num_gpu_blocks( memory_fraction_value=memory_fraction_value, tp_size=tp_size, pp_size=pp_size, - attention_dp_size=( - attention_dp_size if attention_dp_size is not None else 1 - ), + attention_dp_size=(attention_dp_size if attention_dp_size is not None else 1), moe_tp_size=moe_tp_size, moe_ep_size=moe_ep_size, gemm_quant_mode=_quant_mode_name("gemm", gemm_dtype), @@ -215,9 +202,7 @@ def estimate_kv_bytes_per_token( ) value = estimator.kv_bytes_per_token() if value is None or value <= 0: - raise ValueError( - f"could not derive KV bytes per token for model {model_name!r}" - ) + raise ValueError(f"could not derive KV bytes per token for model {model_name!r}") return int(value) @@ -228,9 +213,7 @@ def resolve_model_context_length(model_name: str) -> int: from aiconfigurator_core.sdk.memory import NaiveKVCacheEstimator # Both modules ship in the same distribution; reuse the canonical loader. - config = NaiveKVCacheEstimator._load_config( - model_name, allow_hf_config_download=True - ) + config = NaiveKVCacheEstimator._load_config(model_name, allow_hf_config_download=True) if not isinstance(config, dict): raise ValueError(f"could not load Hugging Face config for {model_name!r}") text_config = config.get("text_config") @@ -248,9 +231,7 @@ def resolve_model_context_length(model_name: str) -> int: value = mapping.get(name) if isinstance(value, int) and not isinstance(value, bool) and value > 0: return value - raise ValueError( - f"Hugging Face config for {model_name!r} does not declare a maximum context length" - ) + raise ValueError(f"Hugging Face config for {model_name!r} does not declare a maximum context length") def _resolve_block_size(raw: dict[str, Any], backend: str) -> int: @@ -288,6 +269,4 @@ def _quant_mode_name(field: str, value: str | None) -> str | None: return enum_cls[normalized].name except KeyError: allowed = ", ".join(member.name for member in enum_cls) - raise ValueError( - f"unsupported AIC {field} quant mode {value!r}; supported values: {allowed}" - ) from None + raise ValueError(f"unsupported AIC {field} quant mode {value!r}; supported values: {allowed}") from None diff --git a/python/aisimulate/src/aisimulate/sweeper/__init__.py b/python/aisimulate/src/aisimulate/sweeper/__init__.py index 1a97a86e3..f85fc97be 100644 --- a/python/aisimulate/src/aisimulate/sweeper/__init__.py +++ b/python/aisimulate/src/aisimulate/sweeper/__init__.py @@ -16,6 +16,8 @@ from .config import ( AdapterSearchConfig, Candidate, + DatabaseMode, + ForwardModel, OptimizationGoal, OptimizationTarget, SearchSpace, @@ -45,6 +47,7 @@ from .replay import ( REPLAY_SPEC_API_VERSION, BackendDeploymentSpec, + ForwardPassEstimatorSpec, HookCapability, ReplayOutputRequirements, ReplayReport, @@ -71,6 +74,8 @@ _LAZY_EXPORTS = { "build_backend_deployment": (".deploy", "build_backend_deployment"), + "ForwardPassEstimatorResolver": (".forward_pass_estimator", "ForwardPassEstimatorResolver"), + "ForwardPassEstimatorResolutionError": (".forward_pass_estimator", "ForwardPassEstimatorResolutionError"), "NoPerfDatabase": (".kv_estimate", "NoPerfDatabase"), "estimate_kv_tokens": (".kv_estimate", "estimate_kv_tokens"), "feasible_shape_tokens": (".kv_estimate", "feasible_shape_tokens"), @@ -131,7 +136,12 @@ def __getattr__(name: str) -> Any: "CandidateRetention", "CandidateStatus", "ConditionalSearchSpace", + "DatabaseMode", "DisaggParallelConfig", + "ForwardModel", + "ForwardPassEstimatorResolutionError", + "ForwardPassEstimatorResolver", + "ForwardPassEstimatorSpec", "HookCapability", "InfeasibleCandidate", "ModelHardware", diff --git a/python/aisimulate/src/aisimulate/sweeper/config.py b/python/aisimulate/src/aisimulate/sweeper/config.py index 2431e649e..910aafde6 100644 --- a/python/aisimulate/src/aisimulate/sweeper/config.py +++ b/python/aisimulate/src/aisimulate/sweeper/config.py @@ -58,6 +58,34 @@ def maximize(self) -> bool: return self not in {OptimizationTarget.TTFT, OptimizationTarget.E2E_LATENCY} +class DatabaseMode(str, Enum): + """Performance-data source policy used by the forward-pass estimator. + + ``SILICON`` uses collected data, ``HYBRID`` falls back from collected data + to the calibrated empirical model, ``EMPIRICAL`` always uses that calibrated + model, and ``SOL`` uses the uncalibrated analytic speed-of-light estimate. + """ + + SILICON = "SILICON" + HYBRID = "HYBRID" + EMPIRICAL = "EMPIRICAL" + SOL = "SOL" + + +class ForwardModel(str, Enum): + """Forward-pass model used for every candidate in a sweep.""" + + OP_LEVEL = "op_level" + FPM = "fpm" + + +class ForwardPassFallbackPolicy(str, Enum): + """Behavior when Core cannot construct the requested native estimator.""" + + ERROR = "error" + REGRESSION = "regression" + + class SLATarget(BaseModel): """Latency bounds in milliseconds. @@ -388,7 +416,6 @@ class SearchSpace(BaseModel): # deployment: branch + backend + legal parallel shapes deployment_mode: list[str] = ["disagg", "agg"] # branches to explore; pin with one backend: list[str] = ["vllm"] # vllm | sglang | trtllm - backend_version: str | None = None parallel_configs: list[dict[str, Any]] = Field(default_factory=list) # generated when empty parallel_configs_by_mode: dict[str, list[dict[str, Any]]] = Field(default_factory=dict) flat_parallel_modes: list[str] = Field(default_factory=list) @@ -398,6 +425,19 @@ class SearchSpace(BaseModel): # pinned model_name: str # HF id or private model name hardware_sku: str # e.g. "h200_sxm" + # A string pins the sole configured backend. A mapping pins backends + # independently while leaving omitted backends on latest-version resolution. + backend_version: str | dict[str, str] | None = None + database_mode: DatabaseMode = DatabaseMode.SILICON + # Core owns the fixed transfer-kind vocabulary and resolves this request to + # a canonical explicit policy before search. None means Core's default (all). + transfer_policy: str | list[str] | None = None + forward_model: ForwardModel = ForwardModel.OP_LEVEL + forward_pass_fallback_policy: ForwardPassFallbackPolicy = ForwardPassFallbackPolicy.ERROR + forward_pass_options: dict[str, Any] | None = None + # Request-scoped system-definition/data roots. ``default`` expands to the + # packaged AISimulate Core systems directory without mutating process globals. + systems_paths: list[str] = Field(default_factory=lambda: ["default"]) gpu_budget: int = 32 # max GPUs per candidate min_gpu_budget: int | None = None context_length: int | None = None @@ -444,6 +484,31 @@ class SearchSpace(BaseModel): engine_log_discrete: list[str] = Field(default_factory=list) engine_integer_log_ranges: dict[str, list[int]] = Field(default_factory=dict) + @field_validator("database_mode", mode="before") + @classmethod + def _normalize_database_mode(cls, value: Any) -> Any: + return value.upper() if isinstance(value, str) else value + + @field_validator("forward_model", mode="before") + @classmethod + def _normalize_lowercase_control(cls, value: Any) -> Any: + return value.lower() if isinstance(value, str) else value + + @field_validator("systems_paths", mode="before") + @classmethod + def _normalize_systems_paths(cls, value: Any) -> Any: + if isinstance(value, str): + value = [part.strip() for part in value.split(",") if part.strip()] + return value + + @field_validator("systems_paths") + @classmethod + def _validate_systems_paths(cls, value: list[str]) -> list[str]: + cleaned = [entry.strip() for entry in value if entry.strip()] + if not cleaned: + raise ValueError("systems_paths must contain at least one path or 'default'") + return list(dict.fromkeys(cleaned)) + @model_validator(mode="after") def _validate_search_choices(self) -> SearchSpace: """Every backend dimension is a non-empty subset of its allowed choices.""" @@ -469,6 +534,41 @@ def _validate_search_choices(self) -> SearchSpace: raise ValueError(f"{field_name} must contain positive integers") return self + @model_validator(mode="after") + def _validate_backend_versions(self) -> SearchSpace: + configured = list(dict.fromkeys(self.backend)) + if isinstance(self.backend_version, str): + if not self.backend_version.strip(): + raise ValueError("backend_version must be a non-empty version") + if len(configured) != 1: + raise ValueError( + "a string backend_version requires exactly one configured backend; " + "use a {backend: version} mapping for a multi-backend search" + ) + self.backend_version = self.backend_version.strip() + elif isinstance(self.backend_version, dict): + unknown = sorted(set(self.backend_version) - set(configured)) + if unknown: + raise ValueError(f"backend_version contains unconfigured backend(s): {unknown}") + invalid = [ + backend + for backend, version in self.backend_version.items() + if not isinstance(version, str) or not version.strip() + ] + if invalid: + raise ValueError(f"backend_version needs a non-empty version for {sorted(invalid)}") + self.backend_version = {backend: version.strip() for backend, version in self.backend_version.items()} + return self + + def requested_backend_version(self, backend: str) -> str | None: + """Return the version pin for ``backend``; ``None`` means resolve latest.""" + + if isinstance(self.backend_version, str): + return self.backend_version + if isinstance(self.backend_version, dict): + return self.backend_version.get(backend) + return None + @model_validator(mode="after") def _validate_gpu_budget(self) -> SearchSpace: """A minimum GPU budget must be positive and within the maximum budget.""" diff --git a/python/aisimulate/src/aisimulate/sweeper/deploy.py b/python/aisimulate/src/aisimulate/sweeper/deploy.py index 6d808de95..9fec48370 100644 --- a/python/aisimulate/src/aisimulate/sweeper/deploy.py +++ b/python/aisimulate/src/aisimulate/sweeper/deploy.py @@ -5,10 +5,11 @@ from __future__ import annotations +from collections.abc import Mapping from typing import Any from ..aic import estimate_kv_bytes_per_token, materialize_aic_num_gpu_blocks -from .replay import BackendDeploymentSpec +from .replay import BackendDeploymentSpec, ForwardPassEstimatorSpec def _role_prefix(role: str) -> str: @@ -16,26 +17,47 @@ def _role_prefix(role: str) -> str: return "" if role == "agg" else f"{role}_" -def _performance_model_metadata(sample: dict[str, Any], role: str, *, backend_version: str) -> dict[str, Any]: - """Keep optional perf-model identity separate from runtime timing args.""" - prefix = _role_prefix(role) - moe_tp = int(sample[f"{prefix}moe_tp"]) - moe_ep = int(sample[f"{prefix}moe_ep"]) - config: dict[str, Any] = { - "backend": sample["backend"], - "backend_version": backend_version, - "system": sample["hardware_sku"], - "model_path": sample["model_name"], - "tp_size": int(sample[f"{prefix}tp"]), - "attention_dp_size": int(sample[f"{prefix}attention_dp"]), - "moe_tp_size": moe_tp if moe_tp * moe_ep > 1 else None, - "moe_ep_size": moe_ep if moe_tp * moe_ep > 1 else None, - "nextn": sample.get("aic_nextn"), +def _performance_model_metadata( + sample: dict[str, Any], + role: str, + *, + backend_version: str, + forward_pass_estimator: ForwardPassEstimatorSpec | None, +) -> dict[str, Any]: + """Preserve the exact role config consumed by Replay plus Core provenance.""" + if forward_pass_estimator is None: + prefix = _role_prefix(role) + moe_tp = int(sample[f"{prefix}moe_tp"]) + moe_ep = int(sample[f"{prefix}moe_ep"]) + return { + "provider": "aic", + "config": { + "backend": sample["backend"], + "backend_version": backend_version, + "system": sample["hardware_sku"], + "model_path": sample["model_name"], + "tp_size": int(sample[f"{prefix}tp"]), + "attention_dp_size": int(sample[f"{prefix}attention_dp"]), + "moe_tp_size": moe_tp if moe_tp * moe_ep > 1 else None, + "moe_ep_size": moe_ep if moe_tp * moe_ep > 1 else None, + "nextn": sample.get("aic_nextn"), + }, + } + return { + "provider": "aic", + "config": dict(forward_pass_estimator.config), + "options": forward_pass_estimator.options, + "selection": forward_pass_estimator.diagnostics, } - return {"provider": "aic", "config": config} -def _engine_args_payload(sample: dict[str, Any], role: str, *, backend_version: str) -> dict[str, Any]: +def _engine_args_payload( + sample: dict[str, Any], + role: str, + *, + backend_version: str, + forward_pass_estimator: ForwardPassEstimatorSpec | None, +) -> dict[str, Any]: """Build the runner-neutral engine argument payload for one role.""" prefix = _role_prefix(role) tp = int(sample[f"{prefix}tp"]) @@ -69,6 +91,8 @@ def _engine_args_payload(sample: dict[str, Any], role: str, *, backend_version: memory_fraction_field: float(memory_fraction), "enable_prefix_caching": bool(sample[f"{role}_enable_prefix_caching"]), } + if forward_pass_estimator is not None and forward_pass_estimator.performance_data_root: + payload["systems_path"] = forward_pass_estimator.performance_data_root if backend == "vllm" and sample.get("context_length") is not None: payload["max_model_len"] = int(sample["context_length"]) if moe_tp * moe_ep > 1: @@ -84,8 +108,21 @@ def _engine_args_payload(sample: dict[str, Any], role: str, *, backend_version: if sample.get(f"{role}_num_gpu_blocks") is not None: payload["num_gpu_blocks"] = int(sample[f"{role}_num_gpu_blocks"]) payload.pop(memory_fraction_field, None) - if sample.get(f"{role}_timing_model") is not None: - payload["timing_model"] = dict(sample[f"{role}_timing_model"]) + authored_timing_model = sample.get(f"{role}_timing_model") + if authored_timing_model is None and forward_pass_estimator is not None: + timing_config = dict(forward_pass_estimator.config) + if forward_pass_estimator.options is not None: + timing_config["options"] = dict(forward_pass_estimator.options) + payload["timing_model"] = { + "type": "external", + "provider": "aic", + "config": timing_config, + } + for name in tuple(payload): + if (name.startswith("aic_") and name != "aic_nextn") or name == "systems_path": + payload.pop(name, None) + elif authored_timing_model is not None: + payload["timing_model"] = dict(authored_timing_model) if sample.get(f"{role}_num_gpu_blocks") is None: payload = materialize_aic_num_gpu_blocks(payload) for name in ( @@ -117,13 +154,32 @@ def _engine_args_payload(sample: dict[str, Any], role: str, *, backend_version: return payload -def build_backend_deployment(sample: dict[str, Any], *, backend_version: str) -> BackendDeploymentSpec: +def build_backend_deployment( + sample: dict[str, Any], + *, + backend_version: str, + forward_pass_estimators: Mapping[str, ForwardPassEstimatorSpec] | None = None, +) -> BackendDeploymentSpec: """Build the Dynamo-independent backend part of a :class:`ReplaySpec`.""" mode = sample["deployment_mode"] + roles = ("agg",) if mode == "agg" else ("prefill", "decode") + resolved_estimators = dict(forward_pass_estimators or {}) + if resolved_estimators and set(resolved_estimators) != set(roles): + raise ValueError( + f"forward-pass estimators must exactly match candidate roles {roles}, got {sorted(resolved_estimators)}" + ) + for role, estimator in resolved_estimators.items(): + if estimator.backend != sample["backend"] or estimator.backend_version != backend_version: + raise ValueError( + "forward-pass estimator identity does not match the sampled backend/version: " + f"role={role}, {estimator.backend}/{estimator.backend_version} != " + f"{sample['backend']}/{backend_version}" + ) common = { "deployment_mode": mode, "backend": sample["backend"], "backend_version": backend_version, + "forward_pass_estimators": resolved_estimators, "parallel_config": { key: value for key, value in sample.items() @@ -154,19 +210,37 @@ def build_backend_deployment(sample: dict[str, Any], *, backend_version: str) -> }, "performance_model_metadata": { ("aggregated" if role == "agg" else role): _performance_model_metadata( - sample, role, backend_version=backend_version + sample, + role, + backend_version=backend_version, + forward_pass_estimator=resolved_estimators.get(role), ) - for role in (("agg",) if mode == "agg" else ("prefill", "decode")) + for role in roles }, } if mode == "agg": return BackendDeploymentSpec( - agg_engine_args=_engine_args_payload(sample, "agg", backend_version=backend_version), + agg_engine_args=_engine_args_payload( + sample, + "agg", + backend_version=backend_version, + forward_pass_estimator=resolved_estimators.get("agg"), + ), num_workers=int(sample["replicas"]), **common, ) - prefill_args = _engine_args_payload(sample, "prefill", backend_version=backend_version) - decode_args = _engine_args_payload(sample, "decode", backend_version=backend_version) + prefill_args = _engine_args_payload( + sample, + "prefill", + backend_version=backend_version, + forward_pass_estimator=resolved_estimators.get("prefill"), + ) + decode_args = _engine_args_payload( + sample, + "decode", + backend_version=backend_version, + forward_pass_estimator=resolved_estimators.get("decode"), + ) if sample.get("kv_transfer_bytes_per_token") == "auto": resolved = max( int(prefill_args["kv_bytes_per_token"]), diff --git a/python/aisimulate/src/aisimulate/sweeper/forward_pass_estimator.py b/python/aisimulate/src/aisimulate/sweeper/forward_pass_estimator.py new file mode 100644 index 000000000..284079674 --- /dev/null +++ b/python/aisimulate/src/aisimulate/sweeper/forward_pass_estimator.py @@ -0,0 +1,175 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Resolve exact per-role Sweeper forward-pass estimator contracts.""" + +from __future__ import annotations + +import os +from importlib import resources +from typing import Any + +from aiconfigurator_core.sdk import ( + ForwardPassPerfModelConfig, + ForwardPassPerfOptions, + RustForwardPassPerfModel, +) + +from .config import SearchSpace +from .replay import ForwardPassEstimatorSpec + + +class ForwardPassEstimatorResolutionError(ValueError): + """A configured forward-pass estimator identity cannot be resolved exactly.""" + + +def resolve_systems_paths(configured: list[str]) -> tuple[str, ...]: + """Expand and validate request-scoped system roots without setting globals.""" + + packaged = os.fspath(resources.files("aiconfigurator_core") / "systems") + resolved: list[str] = [] + for entry in configured: + path = packaged if entry.lower() == "default" else os.path.abspath(os.path.expanduser(entry)) + if not os.path.isdir(path): + raise ForwardPassEstimatorResolutionError(f"systems_paths entry is not an existing directory: {entry!r}") + if path not in resolved: + resolved.append(path) + return tuple(resolved) + + +_DEFAULT_BLOCK_SIZE = {"vllm": 64, "sglang": 1, "trtllm": 32} + + +def _role_prefix(role: str) -> str: + return "" if role == "agg" else f"{role}_" + + +class ForwardPassEstimatorResolver: + """Resolve and cache Core-owned estimator identities for concrete roles. + + The search-space request alone is not an estimator identity: whole-forward + FPM cells include topology and backend block size. ``resolve_candidate`` is + called after a suggestion has been unrolled but before its ``ReplaySpec`` + reaches a runner. Every exact role request crosses the canonical Core + constructor, and the returned config/provenance is carried unchanged. + """ + + def __init__(self, search_space: SearchSpace) -> None: + self._search_space = search_space + self._systems_paths = resolve_systems_paths(search_space.systems_paths) + raw_options = search_space.forward_pass_options + try: + self._options = None if raw_options is None else ForwardPassPerfOptions(**raw_options) + except TypeError as exc: + raise ForwardPassEstimatorResolutionError(f"invalid forward_pass_options: {exc}") from exc + self._resolved: dict[str, ForwardPassEstimatorSpec] = {} + + def _request(self, sample: dict[str, Any], role: str) -> ForwardPassPerfModelConfig: + backend = str(sample["backend"]) + prefix = _role_prefix(role) + moe_tp = int(sample[f"{prefix}moe_tp"]) + moe_ep = int(sample[f"{prefix}moe_ep"]) + block_size = sample[f"{role}_block_size"] + if block_size is None: + block_size = _DEFAULT_BLOCK_SIZE[backend] + transfer_policy: Any = self._search_space.transfer_policy + if isinstance(transfer_policy, list): + transfer_policy = tuple(transfer_policy) + nextn = sample.get("aic_nextn") + if nextn is None: + nextn = self._search_space.aic_nextn + return ForwardPassPerfModelConfig( + model=self._search_space.model_name, + system=self._search_space.hardware_sku, + backend=backend, + backend_version=self._search_space.requested_backend_version(backend), + tp=int(sample[f"{prefix}tp"]), + pp=int(sample[f"{prefix}pp"]), + attention_dp=int(sample[f"{prefix}attention_dp"]), + moe_tp_size=moe_tp if moe_tp * moe_ep > 1 else None, + moe_ep_size=moe_ep if moe_tp * moe_ep > 1 else None, + nextn=int(nextn or 0), + kv_block_size=int(block_size), + forward_model=self._search_space.forward_model.value, + database_mode=self._search_space.database_mode.value, + transfer_policy=transfer_policy, + systems_paths=self._systems_paths, + fallback_policy=self._search_space.forward_pass_fallback_policy.value, + ) + + def _resolve(self, request: ForwardPassPerfModelConfig, role: str) -> ForwardPassEstimatorSpec: + request_payload = vars(request) + cache_key = repr(request) + cached = self._resolved.get(cache_key) + if cached is not None: + return cached + + model: RustForwardPassPerfModel | None = None + try: + model = RustForwardPassPerfModel.best_available(request, self._options) + diagnostics = model.diagnostics() + except Exception as exc: + raise ForwardPassEstimatorResolutionError( + "Core cannot construct the exact forward-pass estimator for " + f"role={role}, model={request.model}, system={request.system}, " + f"backend={request.backend}, tp={request.tp}, pp={request.pp}, " + f"attention_dp={request.attention_dp}, moe_tp={request.moe_tp_size}, " + f"moe_ep={request.moe_ep_size}, kv_block_size={request.kv_block_size}: {exc}" + ) from exc + finally: + if model is not None: + model.close() + + provenance = diagnostics.get("provenance") + if not isinstance(provenance, dict) or not isinstance(provenance.get("config"), dict): + raise ForwardPassEstimatorResolutionError( + f"Core returned no resolved provenance for {request.system}/{request.backend}/{role}" + ) + resolved_config = dict(provenance["config"]) + if not resolved_config.get("backend_version"): + raise ForwardPassEstimatorResolutionError( + f"Core did not resolve an exact backend version for {request.system}/{request.backend}/{role}" + ) + selected_root = provenance.get("selected_systems_root") + if selected_root and resolved_config.get("systems_paths") != [str(selected_root)]: + raise ForwardPassEstimatorResolutionError( + "Core provenance must pin its selected systems root in the resolved config: " + f"config={resolved_config.get('systems_paths')!r}, selected={selected_root!r}" + ) + for field in ( + "model", + "system", + "backend", + "tp", + "pp", + "attention_dp", + "moe_tp_size", + "moe_ep_size", + "nextn", + "kv_block_size", + "forward_model", + ): + if resolved_config.get(field) != request_payload.get(field): + raise ForwardPassEstimatorResolutionError( + f"Core changed exact candidate field {field!r}: " + f"requested={request_payload.get(field)!r}, resolved={resolved_config.get(field)!r}" + ) + spec = ForwardPassEstimatorSpec( + config=resolved_config, + options=None if self._options is None else self._options.to_dict(), + diagnostics=diagnostics, + ) + self._resolved[cache_key] = spec + return spec + + def resolve_candidate(self, sample: dict[str, Any]) -> dict[str, ForwardPassEstimatorSpec]: + """Resolve every concrete engine role before replay trial execution.""" + + roles = ("agg",) if sample["deployment_mode"] == "agg" else ("prefill", "decode") + resolved = {role: self._resolve(self._request(sample, role), role) for role in roles} + versions = {spec.backend_version for spec in resolved.values()} + if len(versions) != 1: + raise ForwardPassEstimatorResolutionError( + f"Core resolved inconsistent backend versions across candidate roles: {sorted(versions)}" + ) + return resolved diff --git a/python/aisimulate/src/aisimulate/sweeper/kv_estimate.py b/python/aisimulate/src/aisimulate/sweeper/kv_estimate.py index 878d6a407..345527ecd 100644 --- a/python/aisimulate/src/aisimulate/sweeper/kv_estimate.py +++ b/python/aisimulate/src/aisimulate/sweeper/kv_estimate.py @@ -29,6 +29,7 @@ from aiconfigurator_core.sdk.memory import estimate_kv_cache from aiconfigurator_core.sdk.perf_database import get_latest_database_version +from .forward_pass_estimator import resolve_systems_paths from .parallel_enum import ParallelShape # Runtime knobs the estimate needs; defaults mirror AIC's memory-estimation tests. @@ -46,9 +47,29 @@ def memory_fraction_kind(backend: str) -> str: return "of_free" if backend == "trtllm" else "of_total" -def resolve_backend_version(hardware_sku: str, backend: str) -> str: - """Latest perf-DB version for the SKU/backend (required by the native estimate).""" - version = get_latest_database_version(hardware_sku, backend) +def resolve_backend_version( + hardware_sku: str, + backend: str, + *, + requested_version: str | None = None, + systems_paths: list[str] | None = None, +) -> str: + """Pinned or latest perf-DB version for the native memory estimate.""" + + resolved_paths = list(resolve_systems_paths(systems_paths)) if systems_paths is not None else None + if requested_version is not None: + from aiconfigurator_core.sdk.perf_database import get_supported_databases + + available = get_supported_databases(systems_paths=resolved_paths) + versions = available.get(hardware_sku, {}).get(backend, []) + if requested_version not in versions: + raise NoPerfDatabase( + f"backend_version={requested_version!r} is unavailable for " + f"hardware_sku={hardware_sku!r}, backend={backend!r}; " + f"available versions: {versions}" + ) + return requested_version + version = get_latest_database_version(hardware_sku, backend, systems_paths=resolved_paths) if version is None: raise NoPerfDatabase( f"no perf database for hardware_sku={hardware_sku!r}, backend={backend!r}; " @@ -64,6 +85,7 @@ def estimate_kv_tokens( hardware_sku: str, backend: str, backend_version: str, + systems_paths: list[str] | None = None, max_num_tokens: int = DEFAULT_MAX_NUM_TOKENS, max_batch_size: int = DEFAULT_MAX_BATCH_SIZE, memory_fraction: float = DEFAULT_MEMORY_FRACTION, @@ -90,6 +112,7 @@ def estimate_kv_tokens( moe_tp_size=shape.moe_tp, moe_ep_size=shape.moe_ep, nextn=nextn, + systems_path=(list(resolve_systems_paths(systems_paths)) if systems_paths is not None else None), allow_naive_fallback=False, ) except ValueError as exc: @@ -114,6 +137,7 @@ def feasible_shape_tokens( backend: str, max_seq_len: int, backend_version: str | None = None, + systems_paths: list[str] | None = None, max_num_tokens: int = DEFAULT_MAX_NUM_TOKENS, max_batch_size: int = DEFAULT_MAX_BATCH_SIZE, memory_fraction: float = DEFAULT_MEMORY_FRACTION, @@ -125,7 +149,7 @@ def feasible_shape_tokens( per distinct shape, so repeated shapes across replica counts are free. """ if backend_version is None: - backend_version = resolve_backend_version(hardware_sku, backend) + backend_version = resolve_backend_version(hardware_sku, backend, systems_paths=systems_paths) feasible: dict[ParallelShape, int] = {} for shape in dict.fromkeys(shapes): # dedup, preserve first-seen order tokens = estimate_kv_tokens( @@ -134,6 +158,7 @@ def feasible_shape_tokens( hardware_sku=hardware_sku, backend=backend, backend_version=backend_version, + systems_paths=systems_paths, max_num_tokens=max_num_tokens, max_batch_size=max_batch_size, memory_fraction=memory_fraction, diff --git a/python/aisimulate/src/aisimulate/sweeper/model_hw.py b/python/aisimulate/src/aisimulate/sweeper/model_hw.py index ee0bdfc65..eedede2a5 100644 --- a/python/aisimulate/src/aisimulate/sweeper/model_hw.py +++ b/python/aisimulate/src/aisimulate/sweeper/model_hw.py @@ -64,7 +64,13 @@ class ModelHardware: max_context: int | None # model's max context length (the default max_seq_len) -def resolve_model_hardware(model_name: str, hardware_sku: str, *, backend: str) -> ModelHardware: +def resolve_model_hardware( + model_name: str, + hardware_sku: str, + *, + backend: str, + systems_paths: list[str] | None = None, +) -> ModelHardware: """Read the model weights + SKU spec (via AIC) to derive is_moe / mla / wideep and the model's max context length.""" model_config = get_model_config_from_model_path(model_name) @@ -74,7 +80,7 @@ def resolve_model_hardware(model_name: str, hardware_sku: str, *, backend: str) mla = is_moe and not allow_pure_tp max_context = model_config.get("context") - system_spec = perf_database.load_system_spec(hardware_sku) + system_spec = perf_database.load_system_spec(hardware_sku, systems_paths=systems_paths) if not system_spec: raise ValueError( f"unknown hardware_sku {hardware_sku!r}: no system config found on AIConfigurator Core's systems path" @@ -114,6 +120,7 @@ def parallel_configs_for( max_batch_size: int = DEFAULT_MAX_BATCH_SIZE, memory_fraction: float = DEFAULT_MEMORY_FRACTION, role_runtime: dict[str, tuple[int, int, float] | tuple[int, int, float, int | None]] | None = None, + systems_paths: list[str] | None = None, ) -> list[ReplicaParallelConfig] | list[DisaggParallelConfig]: """Resolve the model/hardware, then enumerate the parallel configs that fit the GPU budget and can hold a ``max_seq_len``-token sequence. @@ -133,7 +140,12 @@ def parallel_configs_for( :class:`NoViableParallelConfig` when no shape can hold the sequence within the budget. """ - mh = resolve_model_hardware(model_name, hardware_sku, backend=backend) + mh = resolve_model_hardware( + model_name, + hardware_sku, + backend=backend, + systems_paths=systems_paths, + ) seq_len = max_seq_len if max_seq_len is not None else mh.max_context if seq_len is None: raise ValueError(f"max_seq_len is required: {model_name} config exposes no max context length") @@ -177,6 +189,7 @@ def feasible_for(role: str, shapes): hardware_sku=hardware_sku, backend=backend, backend_version=backend_version, + systems_paths=systems_paths, max_seq_len=seq_len, max_num_tokens=role_tokens, max_batch_size=role_batch, diff --git a/python/aisimulate/src/aisimulate/sweeper/replay.py b/python/aisimulate/src/aisimulate/sweeper/replay.py index 37970d4cc..183f36981 100644 --- a/python/aisimulate/src/aisimulate/sweeper/replay.py +++ b/python/aisimulate/src/aisimulate/sweeper/replay.py @@ -19,6 +19,59 @@ REPLAY_SPEC_API_VERSION = 1 +@dataclass(frozen=True) +class ForwardPassEstimatorSpec: + """Resolved output of Core's canonical forward-pass constructor. + + The config is the sole estimator identity carried by Sweeper and Replay. + Convenience properties below are projections, never independently authored + values. Diagnostics preserve Core's selection result for artifacts. + """ + + config: dict[str, JSONValue] + options: dict[str, JSONValue] | None = None + diagnostics: dict[str, JSONValue] = field(default_factory=dict) + + @property + def model_path(self) -> str: + return str(self.config["model"]) + + @property + def system(self) -> str: + return str(self.config["system"]) + + @property + def backend(self) -> str: + return str(self.config["backend"]) + + @property + def backend_version(self) -> str: + return str(self.config["backend_version"]) + + @property + def database_mode(self) -> str: + return str(self.config["database_mode"]) + + @property + def transfer_policy(self) -> tuple[str, ...]: + return tuple(str(value) for value in self.config.get("transfer_policy") or ()) + + @property + def forward_model(self) -> str: + return str(self.config["forward_model"]) + + @property + def systems_paths(self) -> tuple[str, ...]: + return tuple(str(value) for value in self.config.get("systems_paths") or ()) + + @property + def performance_data_root(self) -> str: + provenance = self.diagnostics.get("provenance") + if isinstance(provenance, dict): + return str(provenance.get("selected_systems_root") or "") + return "" + + @dataclass(frozen=True) class BackendDeploymentSpec: """Concrete backend engines and fleet shape for one candidate.""" @@ -34,6 +87,8 @@ class BackendDeploymentSpec: num_prefill_workers: int = 0 num_decode_workers: int = 0 performance_model_metadata: dict[str, JSONValue] = field(default_factory=dict) + # Appended to preserve the positional constructor slots above. + forward_pass_estimators: dict[str, ForwardPassEstimatorSpec] = field(default_factory=dict) @dataclass(frozen=True) @@ -51,11 +106,7 @@ class ReplaySpec: def runtime_hooks(self) -> tuple[RuntimeHookSpec, ...]: """All requested hooks in deterministic adapter insertion order.""" - return tuple( - hook - for adapter_spec in self.adapters.values() - for hook in adapter_spec.runtime_hooks - ) + return tuple(hook for adapter_spec in self.adapters.values() for hook in adapter_spec.runtime_hooks) @dataclass(frozen=True) @@ -109,8 +160,7 @@ def supports_backend_topology(self, backend: str, topology: str) -> bool: """ return any( - (supported_backend in (backend, "*")) - and (supported_topology in (topology, "*")) + (supported_backend in (backend, "*")) and (supported_topology in (topology, "*")) for supported_backend, supported_topology in self.supported_backend_topologies ) @@ -126,14 +176,10 @@ def supports_attention_dp(self, topology: str, *dp_sizes: int) -> bool: or all(dp_size == 1 for dp_size in dp_sizes) ) - def require_replay_spec_version( - self, api_version: int = REPLAY_SPEC_API_VERSION - ) -> None: + def require_replay_spec_version(self, api_version: int = REPLAY_SPEC_API_VERSION) -> None: """Raise when the runner and Sweeper do not share the replay-spec ABI.""" - versions_are_integers = ( - type(api_version) is int and type(self.replay_spec_api_version) is int - ) + versions_are_integers = type(api_version) is int and type(self.replay_spec_api_version) is int if not versions_are_integers or api_version != self.replay_spec_api_version: raise ValueError( f"ReplaySpec API version {api_version} is incompatible with " @@ -145,21 +191,13 @@ def require_compatible(self, spec: ReplaySpec) -> None: self.require_replay_spec_version(spec.api_version) deployment = spec.backend_deployment - if not self.supports_backend_topology( - deployment.backend, deployment.deployment_mode - ): + if not self.supports_backend_topology(deployment.backend, deployment.deployment_mode): raise ValueError( - f"runner does not support backend/topology " - f"{deployment.backend!r}/{deployment.deployment_mode!r}" + f"runner does not support backend/topology {deployment.backend!r}/{deployment.deployment_mode!r}" ) - unsupported = [ - hook for hook in spec.runtime_hooks if not self.supports_hook(hook) - ] + unsupported = [hook for hook in spec.runtime_hooks if not self.supports_hook(hook)] if unsupported: - labels = ", ".join( - f"{hook.provider}:{hook.kind}@{hook.api_version}" - for hook in unsupported - ) + labels = ", ".join(f"{hook.provider}:{hook.kind}@{hook.api_version}" for hook in unsupported) raise ValueError(f"runner does not support runtime hook(s): {labels}") @@ -199,18 +237,14 @@ def _jsonable(value: Any) -> JSONValue: converted: dict[str, JSONValue] = {} for key, item in value.items(): if not isinstance(key, str): - raise TypeError( - f"canonical replay JSON requires string mapping keys, got {key!r}" - ) + raise TypeError(f"canonical replay JSON requires string mapping keys, got {key!r}") converted[key] = _jsonable(item) return converted if isinstance(value, (list, tuple)): return [_jsonable(item) for item in value] if value is None or isinstance(value, (str, int, float, bool)): return value - raise TypeError( - f"value of type {type(value).__name__} is not supported by replay JSON contracts" - ) + raise TypeError(f"value of type {type(value).__name__} is not supported by replay JSON contracts") def validate_json_value(value: Any, *, path: str = "value") -> None: diff --git a/python/aisimulate/src/aisimulate/sweeper/search.py b/python/aisimulate/src/aisimulate/sweeper/search.py index ae44260cf..5ade9ec2f 100644 --- a/python/aisimulate/src/aisimulate/sweeper/search.py +++ b/python/aisimulate/src/aisimulate/sweeper/search.py @@ -44,7 +44,7 @@ from .config import Candidate, OptimizationGoal, OptimizationTarget, SmartSearchConfig from .deploy import build_backend_deployment from .discovery import resolve_providers -from .kv_estimate import resolve_backend_version +from .forward_pass_estimator import ForwardPassEstimatorResolver from .kv_load import InfeasibleKVCapacity, resolve_kv_load from .provider import ( SEARCH_SPACE_FRAGMENT_API_VERSION, @@ -547,6 +547,7 @@ def _materialize_one( providers: Mapping[str, SweepConfigProvider], provider_plans: Mapping[str, AdapterSearchPlan], runner_factory: RunnerFactory, + forward_pass_estimator_resolver: ForwardPassEstimatorResolver, prediction_config_factory: Callable[[dict[str, Any], ReplaySpec], dict[str, Any]] | None = None, ) -> tuple[_PreparedCandidate | None, _EvalResult | None]: """Build a complete replay specification on the main process.""" @@ -556,13 +557,15 @@ def _materialize_one( selection=selection, parallel_config=parallel_config, ) - backend_version = config.search_space.backend_version or resolve_backend_version( - config.search_space.hardware_sku, selection["backend"] - ) + forward_pass_estimators = forward_pass_estimator_resolver.resolve_candidate(sample) + backend_version = next(iter(forward_pass_estimators.values())).backend_version # The resolved perf-model version is part of the evaluated contract. Keep it # on the candidate so downstream artifact generation cannot independently # select a different backend version. sample["backend_version"] = backend_version + sample["forward_pass_estimators"] = { + role: asdict(estimator) for role, estimator in forward_pass_estimators.items() + } concurrency = config.workload.concurrency workload_payload = config.workload.model_dump(mode="json") if "traffic_load" in selection: @@ -602,7 +605,11 @@ def _materialize_one( # Preserve the concrete load on every candidate, including a fixed absolute # concurrency and one derived from kv_load_ratio. sample["concurrency"] = concurrency - backend_deployment = build_backend_deployment(sample, backend_version=backend_version) + backend_deployment = build_backend_deployment( + sample, + backend_version=backend_version, + forward_pass_estimators=forward_pass_estimators, + ) adapter_specs: dict[str, AdapterReplaySpec] = {} for name, provider in providers.items(): candidate_context = CandidateContext( @@ -1000,6 +1007,11 @@ def run( capabilities = runner_factory.capabilities() capabilities.require_replay_spec_version(REPLAY_SPEC_API_VERSION) + # Parse request-scoped estimator controls once. Exact Core construction is + # deferred until a suggestion has concrete per-role topology and block size, + # then completed before the replay trial reaches a runner. + forward_pass_estimator_resolver = ForwardPassEstimatorResolver(config.search_space) + # Preserve the legacy preflight order: reject an impossible backend/topology # search before adapters perform any potentially expensive preparation. branches = enumerate_branches( @@ -1393,6 +1405,7 @@ def _run_branch_round(state: _BranchSearchState) -> None: providers=resolved_providers, provider_plans=provider_plans, runner_factory=runner_factory, + forward_pass_estimator_resolver=forward_pass_estimator_resolver, prediction_config_factory=prediction_config_factory, ) if build_result is not None: diff --git a/python/aisimulate/src/aisimulate/sweeper/search_space.py b/python/aisimulate/src/aisimulate/sweeper/search_space.py index 4fd51162e..335b5c74e 100644 --- a/python/aisimulate/src/aisimulate/sweeper/search_space.py +++ b/python/aisimulate/src/aisimulate/sweeper/search_space.py @@ -243,16 +243,22 @@ def matches_custom(config: _ParallelConfig) -> bool: ): continue try: + forward_pass_estimator_kwargs: dict[str, Any] = {} + requested_version = ss.requested_backend_version(backend) + if requested_version is not None: + forward_pass_estimator_kwargs["backend_version"] = requested_version + if ss.systems_paths != ["default"]: + forward_pass_estimator_kwargs["systems_paths"] = ss.systems_paths legal = parallel_configs_for( ss.model_name, ss.hardware_sku, gpu_budget=ss.gpu_budget, deployment_mode=deployment_mode, backend=backend, - backend_version=ss.backend_version, min_gpu_budget=ss.min_gpu_budget, max_seq_len=max_seq_len, role_runtime=role_runtime(backend, deployment_mode), + **forward_pass_estimator_kwargs, ) except (NoPerfDatabase, NoViableParallelConfig): continue # backend unusable for this mode -> drop it from the search diff --git a/python/aisimulate/tests/unit/sdk/test_core_namespace.py b/python/aisimulate/tests/unit/sdk/test_core_namespace.py index f08797b3d..afb2f63b5 100644 --- a/python/aisimulate/tests/unit/sdk/test_core_namespace.py +++ b/python/aisimulate/tests/unit/sdk/test_core_namespace.py @@ -16,3 +16,9 @@ def test_task_v2_remains_upper_owned() -> None: assert task.__module__ == "aiconfigurator.sdk.task_v2" assert importlib.util.find_spec("aiconfigurator_core.sdk.task_v2") is None + + +def test_low_level_engine_builder_is_not_a_python_estimator_api() -> None: + core = importlib.import_module("aiconfigurator_core") + + assert not hasattr(core, "AicEngineBuilder") diff --git a/python/aisimulate/tests/unit/sdk/test_rust_engine_step.py b/python/aisimulate/tests/unit/sdk/test_rust_engine_step.py index fa412d86b..c7407a641 100644 --- a/python/aisimulate/tests/unit/sdk/test_rust_engine_step.py +++ b/python/aisimulate/tests/unit/sdk/test_rust_engine_step.py @@ -723,27 +723,20 @@ def test_nemotron_super_fp8_native_estimation_uses_packaged_moe_data() -> None: pytest.importorskip("aiconfigurator_core") from aiconfigurator_core.sdk.rust_engine_step import RustForwardPassPerfModel - model = RustForwardPassPerfModel.from_native( + model = RustForwardPassPerfModel.best_available( { - "schema_version": 1, - "model_name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", - "system_name": "h100_sxm", + "model": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", + "system": "h100_sxm", "backend": "vllm", "backend_version": "0.24.0", - "tp_size": 4, - "pp_size": 1, - "attention_dp_size": 1, - "cp_size": None, + "tp": 4, + "pp": 1, + "attention_dp": 1, "moe_tp_size": 1, "moe_ep_size": 4, - "weight_dtype": "fp8", - "activation_dtype": None, - "moe_dtype": None, - "kv_cache_dtype": None, + "gemm_quant_mode": "fp8", "kv_block_size": 16, - "nextn": None, - "nextn_accept_rates": None, - "extra": {}, + "fallback_policy": "error", }, { "bucket_count": 16, @@ -798,25 +791,19 @@ def test_forward_pass_perf_model_native_default_directional_bounds_end_to_end() from aiconfigurator.sdk.rust_engine_step import RustForwardPassPerfModel config = { - "schema_version": 1, - "model_name": "Qwen/Qwen3-32B", - "system_name": "h200_sxm", + "model": "Qwen/Qwen3-32B", + "system": "h200_sxm", "backend": "trtllm", "backend_version": "1.3.0rc10", - "tp_size": 4, - "pp_size": 1, + "tp": 4, + "pp": 1, "moe_tp_size": None, "moe_ep_size": None, - "attention_dp_size": 1, - "weight_dtype": None, - "moe_dtype": None, - "activation_dtype": None, - "kv_cache_dtype": None, + "attention_dp": 1, "kv_block_size": None, - "nextn": None, - "extra": {}, + "fallback_policy": "error", } - model = RustForwardPassPerfModel.from_native( + model = RustForwardPassPerfModel.best_available( config, { "min_observations": 2, @@ -899,23 +886,17 @@ def test_forward_pass_perf_model_best_available_falls_back_on_bad_config() -> No from aiconfigurator.sdk.rust_engine_step import RustForwardPassPerfModel config = { - "schema_version": 1, - "model_name": "this/model-does-not-exist-xyz", - "system_name": "h200_sxm", + "model": "this/model-does-not-exist-xyz", + "system": "h200_sxm", "backend": "trtllm", "backend_version": "1.3.0rc10", - "tp_size": 1, - "pp_size": 1, + "tp": 1, + "pp": 1, "moe_tp_size": None, "moe_ep_size": None, - "attention_dp_size": 1, - "weight_dtype": None, - "moe_dtype": None, - "activation_dtype": None, - "kv_cache_dtype": None, + "attention_dp": 1, "kv_block_size": None, - "nextn": None, - "extra": {}, + "fallback_policy": "regression", } model = RustForwardPassPerfModel.best_available(config, {"min_observations": 2}) diag = model.diagnostics() diff --git a/tests/sweeper/test_config.py b/tests/sweeper/test_config.py index 4876b29f2..903e3a5c7 100644 --- a/tests/sweeper/test_config.py +++ b/tests/sweeper/test_config.py @@ -10,6 +10,8 @@ from aisimulate.recommend import recommendation_to_sweeper from aisimulate.sweeper import OptimizationTarget, SmartSearchConfig from aisimulate.sweeper.config import ( + DatabaseMode, + ForwardModel, OptimizationGoal, SearchSpace, SLATarget, @@ -133,12 +135,71 @@ def test_defaults_are_backend_only(): assert config.adapters == {} assert config.goal.target is OptimizationTarget.THROUGHPUT assert config.sweep.parallel_evals == 16 + assert config.search_space.database_mode is DatabaseMode.SILICON + assert config.search_space.transfer_policy is None + assert config.search_space.forward_model is ForwardModel.OP_LEVEL + assert config.search_space.systems_paths == ["default"] dumped = config.search_space.model_dump() assert ( - not {"planner_scaling_policy", "router_mode", "num_g2_blocks"} & dumped.keys() + not { + "engine_step_backend", + "planner_scaling_policy", + "router_mode", + "num_g2_blocks", + } + & dumped.keys() ) +def test_forward_pass_estimator_request_controls_preserve_policy_for_resolution(): + search = SearchSpace( + **_search_space( + backend=["vllm"], + backend_version=" 0.11.0 ", + database_mode="hybrid", + transfer_policy="balanced,xop", + forward_model="FPM", + systems_paths="default, /tmp/custom-systems", + ) + ) + + assert search.requested_backend_version("vllm") == "0.11.0" + assert search.database_mode is DatabaseMode.HYBRID + assert search.transfer_policy == "balanced,xop" + assert search.forward_model is ForwardModel.FPM + assert search.systems_paths == ["default", "/tmp/custom-systems"] + + +def test_backend_version_mapping_pins_backends_independently(): + search = SearchSpace( + **_search_space( + backend=["vllm", "sglang"], + backend_version={"vllm": "0.11.0"}, + ) + ) + + assert search.requested_backend_version("vllm") == "0.11.0" + assert search.requested_backend_version("sglang") is None + + +@pytest.mark.parametrize( + ("overrides", "message"), + [ + ( + {"backend": ["vllm", "sglang"], "backend_version": "0.11.0"}, + "mapping", + ), + ({"backend_version": {"sglang": "0.5.6"}}, "unconfigured"), + ({"forward_model": "formula"}, "forward_model"), + ({"engine_step_backend": "rust"}, "engine_step_backend"), + ({"systems_paths": []}, "systems_paths"), + ], +) +def test_invalid_forward_pass_estimator_controls_fail_in_schema(overrides, message): + with pytest.raises(ValidationError, match=message): + SearchSpace(**_search_space(**overrides)) + + def test_extra_fields_are_forbidden_at_each_boundary(): with pytest.raises(ValidationError, match="Extra inputs are not permitted"): SmartSearchConfig( diff --git a/tests/sweeper/test_deploy.py b/tests/sweeper/test_deploy.py index a55c14ff6..d391fe8fb 100644 --- a/tests/sweeper/test_deploy.py +++ b/tests/sweeper/test_deploy.py @@ -13,6 +13,7 @@ ParallelShape, ReplicaParallelConfig, ) +from aisimulate.sweeper.replay import ForwardPassEstimatorSpec from aisimulate.sweeper.sample import unroll_sample BACKEND_VERSION = "1.3.0rc10" @@ -178,6 +179,108 @@ def test_optional_backend_runtime_values_are_forwarded(): assert engine["aic_nextn"] == 2 +def test_resolved_forward_pass_estimator_contract_is_preserved_without_leaking_into_engine_args( + monkeypatch, +): + monkeypatch.setattr( + deploy_module, + "materialize_aic_num_gpu_blocks", + lambda payload: {**payload, "num_gpu_blocks": 321}, + ) + forward_pass_estimator = ForwardPassEstimatorSpec( + config={ + "model": "example/model", + "system": "example_sku", + "backend": "trtllm", + "backend_version": BACKEND_VERSION, + "tp": 4, + "pp": 1, + "attention_dp": 1, + "moe_tp_size": 1, + "moe_ep_size": 4, + "nextn": 0, + "kv_block_size": 64, + "database_mode": "HYBRID", + "transfer_policy": ["xshape", "xquant"], + "forward_model": "fpm", + "systems_paths": ["/custom/systems"], + "fallback_policy": "error", + }, + diagnostics={ + "source": "aic", + "provenance": { + "config": { + "model": "example/model", + "system": "example_sku", + "backend": "trtllm", + "backend_version": BACKEND_VERSION, + "tp": 4, + "pp": 1, + "attention_dp": 1, + "moe_tp_size": 1, + "moe_ep_size": 4, + "nextn": 0, + "kv_block_size": 64, + "database_mode": "HYBRID", + "transfer_policy": ["xshape", "xquant"], + "forward_model": "fpm", + "systems_paths": ["/custom/systems"], + "fallback_policy": "error", + }, + "selected_systems_root": "/custom/systems", + }, + }, + ) + sample = unroll_sample( + search_space=_space(), + selection=_agg_selection(), + parallel_config=AGG_MOE, + ) + + deployment = build_backend_deployment( + sample, + backend_version=BACKEND_VERSION, + forward_pass_estimators={"agg": forward_pass_estimator}, + ) + + assert deployment.forward_pass_estimators == {"agg": forward_pass_estimator} + engine = deployment.agg_engine_args + assert engine is not None + assert { + "aic_database_mode", + "aic_transfer_policy", + "aic_forward_model", + "aic_systems_paths", + "aic_engine_step_backend", + }.isdisjoint(engine) + expected_config = { + "model": "example/model", + "system": "example_sku", + "backend": "trtllm", + "backend_version": BACKEND_VERSION, + "tp": 4, + "pp": 1, + "attention_dp": 1, + "moe_tp_size": 1, + "moe_ep_size": 4, + "nextn": 0, + "kv_block_size": 64, + "database_mode": "HYBRID", + "transfer_policy": ["xshape", "xquant"], + "forward_model": "fpm", + "systems_paths": ["/custom/systems"], + "fallback_policy": "error", + } + assert ( + deployment.performance_model_metadata["aggregated"]["config"] == expected_config + ) + assert engine["timing_model"]["config"] == expected_config + assert ( + deployment.performance_model_metadata["aggregated"]["selection"]["source"] + == "aic" + ) + + def test_backend_deployment_contains_no_dynamo_policy_fields(): deployment = _agg_deployment() diff --git a/tests/sweeper/test_forward_pass_estimator.py b/tests/sweeper/test_forward_pass_estimator.py new file mode 100644 index 000000000..ccdc23801 --- /dev/null +++ b/tests/sweeper/test_forward_pass_estimator.py @@ -0,0 +1,264 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Golden Core-owned forward-pass estimator resolution for Sweeper candidates.""" + +from __future__ import annotations + +from dataclasses import asdict + +import pytest + +import aisimulate.sweeper.forward_pass_estimator as forward_pass_estimator_mod +from aisimulate.sweeper.config import SearchSpace +from aisimulate.sweeper.forward_pass_estimator import ( + ForwardPassEstimatorResolver, + ForwardPassEstimatorResolutionError, +) + + +def _space(systems_root, **overrides): + values = { + "model_name": "example/model", + "hardware_sku": "example_system", + "backend": ["vllm"], + "systems_paths": [str(systems_root)], + } + values.update(overrides) + return SearchSpace(**values) + + +def _agg_sample(**overrides): + values = { + "deployment_mode": "agg", + "backend": "vllm", + "tp": 4, + "pp": 1, + "attention_dp": 1, + "moe_tp": 1, + "moe_ep": 4, + "agg_block_size": 16, + } + values.update(overrides) + return values + + +def _disagg_sample(**overrides): + values = { + "deployment_mode": "disagg", + "backend": "vllm", + "prefill_tp": 2, + "prefill_pp": 1, + "prefill_attention_dp": 1, + "prefill_moe_tp": 1, + "prefill_moe_ep": 2, + "prefill_block_size": 16, + "decode_tp": 4, + "decode_pp": 1, + "decode_attention_dp": 1, + "decode_moe_tp": 1, + "decode_moe_ep": 4, + "decode_block_size": 32, + } + values.update(overrides) + return values + + +def _stub_core(monkeypatch, systems_root, *, versions=("0.10.0", "0.11.0")): + calls = [] + + class _Model: + def __init__(self, diagnostics): + self._diagnostics = diagnostics + + def diagnostics(self): + return self._diagnostics + + def close(self): + return None + + class _Core: + @staticmethod + def best_available(config, options): + calls.append((config, options)) + if ( + config.backend_version is not None + and config.backend_version not in versions + ): + raise ValueError( + f"unsupported backend_version {config.backend_version!r}" + ) + if config.transfer_policy == "mystery": + raise ValueError("invalid transfer_policy 'mystery'") + if config.forward_model == "fpm" and config.nextn: + raise ValueError("forward_model='fpm' does not support aic_nextn/MTP") + if config.forward_model == "fpm": + complete = any( + path.name == "fpm_forward_perf.parquet" + and (path.parent / "fpm_forward_perf.metadata.json").is_file() + for path in systems_root.rglob("fpm_forward_perf.parquet") + ) + if not complete: + raise ValueError( + "forward_model='fpm' requires fpm_forward_perf data" + ) + + resolved = asdict(config) + resolved["backend_version"] = config.backend_version or ( + versions[-1] if versions else None + ) + if config.transfer_policy is None: + resolved["transfer_policy"] = ["xshape", "xquant", "xprofile", "xop"] + elif config.transfer_policy == "balanced,xop": + resolved["transfer_policy"] = ["xshape", "xquant", "xop"] + else: + resolved["transfer_policy"] = list(config.transfer_policy) + resolved["systems_paths"] = [str(systems_root)] + return _Model( + { + "source": "aic", + "readiness": "ready", + "provenance": { + "config": resolved, + "selected_systems_root": str(systems_root), + }, + } + ) + + monkeypatch.setattr(forward_pass_estimator_mod, "RustForwardPassPerfModel", _Core) + return calls + + +def test_default_resolution_is_core_owned_concrete_and_reproducible( + monkeypatch, tmp_path +): + calls = _stub_core(monkeypatch, tmp_path) + + spec = ForwardPassEstimatorResolver(_space(tmp_path)).resolve_candidate( + _agg_sample() + )["agg"] + + assert spec.model_path == "example/model" + assert spec.system == "example_system" + assert spec.backend == "vllm" + assert spec.backend_version == "0.11.0" + assert spec.database_mode == "SILICON" + assert spec.transfer_policy == ("xshape", "xquant", "xprofile", "xop") + assert spec.forward_model == "op_level" + assert spec.systems_paths == (str(tmp_path),) + assert spec.performance_data_root == str(tmp_path) + assert calls[0][0].backend_version is None + assert calls[0][0].tp == 4 + assert calls[0][0].moe_ep_size == 4 + assert calls[0][0].kv_block_size == 16 + assert spec.config == spec.diagnostics["provenance"]["config"] + + +def test_pinned_policy_and_options_reach_the_canonical_constructor( + monkeypatch, tmp_path +): + calls = _stub_core(monkeypatch, tmp_path) + search = _space( + tmp_path, + backend_version="0.10.0", + database_mode="HYBRID", + transfer_policy="balanced,xop", + forward_pass_options={"min_observations": 3}, + ) + + spec = ForwardPassEstimatorResolver(search).resolve_candidate(_agg_sample())["agg"] + + request, options = calls[0] + assert request.backend_version == "0.10.0" + assert request.database_mode == "HYBRID" + assert request.transfer_policy == "balanced,xop" + assert options is not None and options.min_observations == 3 + assert spec.transfer_policy == ("xshape", "xquant", "xop") + assert spec.options is not None and spec.options["min_observations"] == 3 + + +def test_invalid_transfer_policy_fails_through_core(monkeypatch, tmp_path): + calls = _stub_core(monkeypatch, tmp_path) + + with pytest.raises(ForwardPassEstimatorResolutionError, match="transfer_policy"): + ForwardPassEstimatorResolver( + _space(tmp_path, transfer_policy="mystery") + ).resolve_candidate(_agg_sample()) + + assert len(calls) == 1 + + +def test_unknown_pinned_version_fails_through_core(monkeypatch, tmp_path): + calls = _stub_core(monkeypatch, tmp_path) + + with pytest.raises( + ForwardPassEstimatorResolutionError, match="unsupported backend_version" + ): + ForwardPassEstimatorResolver( + _space(tmp_path, backend_version="9.9.9") + ).resolve_candidate(_agg_sample()) + + assert len(calls) == 1 + + +def test_fpm_support_is_validated_by_core_before_search(monkeypatch, tmp_path): + _stub_core(monkeypatch, tmp_path) + search = _space(tmp_path, backend_version="0.11.0", forward_model="fpm") + resolver = ForwardPassEstimatorResolver(search) + + with pytest.raises( + ForwardPassEstimatorResolutionError, match="requires fpm_forward_perf" + ): + resolver.resolve_candidate(_agg_sample()) + + version_dir = tmp_path / "data/example_system/dense/vllm/0.11.0" + version_dir.mkdir(parents=True) + (version_dir / "fpm_forward_perf.parquet").touch() + (version_dir / "fpm_forward_perf.metadata.json").write_text("{}") + + spec = resolver.resolve_candidate(_agg_sample())["agg"] + assert spec.forward_model == "fpm" + + +def test_fpm_rejects_mtp_through_core_before_search(monkeypatch, tmp_path): + _stub_core(monkeypatch, tmp_path) + + with pytest.raises( + ForwardPassEstimatorResolutionError, match="does not support aic_nextn" + ): + ForwardPassEstimatorResolver( + _space( + tmp_path, + backend_version="0.11.0", + forward_model="fpm", + aic_nextn=2, + ) + ).resolve_candidate(_agg_sample()) + + +def test_invalid_system_path_fails_concisely(tmp_path): + with pytest.raises( + ForwardPassEstimatorResolutionError, match="not an existing directory" + ): + ForwardPassEstimatorResolver(_space(tmp_path / "missing")) + + +def test_disaggregated_roles_are_resolved_exactly_and_cached(monkeypatch, tmp_path): + calls = _stub_core(monkeypatch, tmp_path) + resolver = ForwardPassEstimatorResolver(_space(tmp_path)) + + first = resolver.resolve_candidate(_disagg_sample()) + second = resolver.resolve_candidate(_disagg_sample()) + + assert second == first + assert len(calls) == 2 + assert first["prefill"].config["tp"] == 2 + assert first["prefill"].config["moe_ep_size"] == 2 + assert first["prefill"].config["kv_block_size"] == 16 + assert first["decode"].config["tp"] == 4 + assert first["decode"].config["moe_ep_size"] == 4 + assert first["decode"].config["kv_block_size"] == 32 + assert all( + spec.config == spec.diagnostics["provenance"]["config"] + for spec in first.values() + ) diff --git a/tests/sweeper/test_model_hw.py b/tests/sweeper/test_model_hw.py index 57a46c8b8..327d19a8c 100644 --- a/tests/sweeper/test_model_hw.py +++ b/tests/sweeper/test_model_hw.py @@ -56,7 +56,7 @@ def test_aic_core_system_spec_contract(monkeypatch): monkeypatch.setattr( mh_mod.perf_database, "load_system_spec", - lambda hardware: { + lambda hardware, systems_paths=None: { "gpu": {"mem_capacity": 80}, "node": {"num_gpus_per_node": 8}, }, diff --git a/tests/sweeper/test_result.py b/tests/sweeper/test_result.py index eb6013683..3985f3eea 100644 --- a/tests/sweeper/test_result.py +++ b/tests/sweeper/test_result.py @@ -21,6 +21,7 @@ CandidateRecord, CandidateRetention, CandidateStatus, + ForwardPassEstimatorSpec, OperationProvenance, ReasonCategory, ReplayReport, @@ -64,6 +65,51 @@ def _config(*, target: str = "throughput") -> SmartSearchConfig: ) +class _StaticResolver: + def __init__(self, specs): + self.specs = specs + + def resolve_candidate(self, sample): + roles = ( + ("agg",) if sample["deployment_mode"] == "agg" else ("prefill", "decode") + ) + return {role: self.specs[sample["backend"]] for role in roles} + + +def _forward_pass_estimator_resolver(search_space): + specs = { + backend: ForwardPassEstimatorSpec( + config={ + "model": search_space.model_name, + "system": search_space.hardware_sku, + "backend": backend, + "backend_version": "1.0", + "database_mode": "SILICON", + "transfer_policy": ["xshape", "xquant", "xprofile", "xop"], + "forward_model": "op_level", + "systems_paths": ["/systems"], + "fallback_policy": "error", + }, + diagnostics={"provenance": {"selected_systems_root": "/systems"}}, + ) + for backend in search_space.backend + } + return _StaticResolver(specs) + + +def _stub_optimizer_dependencies(monkeypatch, branch: BranchSpace) -> None: + monkeypatch.setattr( + search_module, + "enumerate_branches", + lambda *args, **kwargs: [branch], + ) + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) + + def _provenance() -> CandidateProvenance: return CandidateProvenance( model="example/model", @@ -472,16 +518,7 @@ def test_optimizer_guided_run_emits_complete_ledger_and_top_n(monkeypatch): supported_backends={parallel_config: frozenset({"trtllm"})}, knob_choices={"backend": ["trtllm"]}, ) - monkeypatch.setattr( - search_module, - "enumerate_branches", - lambda config, *, max_seq_len=None, runner_capabilities=None: [branch], - ) - monkeypatch.setattr( - search_module, - "resolve_backend_version", - lambda hardware, backend: "1.0", - ) + _stub_optimizer_dependencies(monkeypatch, branch) result = Sweeper( runner_factory=_RunnerFactory(), @@ -507,7 +544,7 @@ def test_optimizer_guided_run_emits_complete_ledger_and_top_n(monkeypatch): assert result.candidates[0].provenance.performance_data[0]["source"] == "parquet" assert result.candidates[0].provenance.performance_data[1]["role"] == "aggregated" assert ( - result.candidates[0].provenance.performance_data[1]["config"]["model_path"] + result.candidates[0].provenance.performance_data[1]["config"]["model"] == "example/model" ) assert result.candidates[0].provenance.power["mean_power_w"] == 400.0 @@ -536,16 +573,7 @@ def test_strict_sla_rejection_is_preserved_in_the_candidate_ledger(monkeypatch): supported_backends={parallel_config: frozenset({"trtllm"})}, knob_choices={"backend": ["trtllm"]}, ) - monkeypatch.setattr( - search_module, - "enumerate_branches", - lambda config, *, max_seq_len=None, runner_capabilities=None: [branch], - ) - monkeypatch.setattr( - search_module, - "resolve_backend_version", - lambda hardware, backend: "1.0", - ) + _stub_optimizer_dependencies(monkeypatch, branch) config_data = _config().model_dump(mode="python") config_data["goal"] = { "target": "throughput", @@ -579,10 +607,7 @@ def test_same_batch_failed_duplicates_are_counted_as_coalesced_hits(monkeypatch) supported_backends={parallel_config: frozenset({"trtllm"})}, knob_choices={"backend": ["trtllm"]}, ) - monkeypatch.setattr( - search_module, "enumerate_branches", lambda *args, **kwargs: [branch] - ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "1.0") + _stub_optimizer_dependencies(monkeypatch, branch) result = Sweeper( runner_factory=_FailingRunnerFactory(), @@ -609,10 +634,7 @@ def test_zero_or_missing_sample_latency_preserves_ranked_sampler_feedback( supported_backends={parallel_config: frozenset({"trtllm"})}, knob_choices={"backend": ["trtllm"]}, ) - monkeypatch.setattr( - search_module, "enumerate_branches", lambda *args, **kwargs: [branch] - ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "1.0") + _stub_optimizer_dependencies(monkeypatch, branch) seen = {} @@ -665,16 +687,7 @@ def test_optimizer_guided_result_separates_unsupported_and_runtime_failure(monke supported_backends={parallel_config: frozenset({"trtllm"})}, knob_choices={"backend": ["trtllm"]}, ) - monkeypatch.setattr( - search_module, - "enumerate_branches", - lambda config, *, max_seq_len=None, runner_capabilities=None: [branch], - ) - monkeypatch.setattr( - search_module, - "resolve_backend_version", - lambda hardware, backend: "1.0", - ) + _stub_optimizer_dependencies(monkeypatch, branch) result = Sweeper( runner_factory=_FailingRunnerFactory(), diff --git a/tests/sweeper/test_search.py b/tests/sweeper/test_search.py index 7985ff43f..01142a600 100644 --- a/tests/sweeper/test_search.py +++ b/tests/sweeper/test_search.py @@ -13,6 +13,7 @@ from aisimulate.sweeper.parallel_enum import ParallelShape, ReplicaParallelConfig from aisimulate.sweeper.replay import ( BackendDeploymentSpec, + ForwardPassEstimatorSpec, ReplayReport, ReplaySpec, RunnerCapabilities, @@ -145,6 +146,34 @@ def _branch(parallel_config): ) +def _forward_pass_estimator_spec(backend="trtllm", version="1.3.0rc10"): + return ForwardPassEstimatorSpec( + config={ + "model": "deepseek-ai/DeepSeek-V3", + "system": "gb200", + "backend": backend, + "backend_version": version, + "database_mode": "SILICON", + "transfer_policy": ["xshape", "xquant", "xprofile", "xop"], + "forward_model": "op_level", + "systems_paths": ["/systems"], + "fallback_policy": "error", + }, + diagnostics={"provenance": {"selected_systems_root": "/systems"}}, + ) + + +class _StaticResolver: + def __init__(self, specs): + self.specs = specs + + def resolve_candidate(self, sample): + roles = ( + ("agg",) if sample["deployment_mode"] == "agg" else ("prefill", "decode") + ) + return {role: self.specs[sample["backend"]] for role in roles} + + def _stub(monkeypatch, branch): monkeypatch.setattr( search_mod, @@ -152,7 +181,14 @@ def _stub(monkeypatch, branch): lambda config, *, max_seq_len=None, runner_capabilities=None: [branch], ) monkeypatch.setattr( - search_mod, "resolve_backend_version", lambda hw, be: "1.3.0rc10" + search_mod, + "ForwardPassEstimatorResolver", + lambda search_space: _StaticResolver( + { + backend: _forward_pass_estimator_spec(backend) + for backend in search_space.backend + } + ), ) @@ -179,13 +215,23 @@ def test_ranks_feasible_best_first_and_passes_replay_specs(monkeypatch): assert all( candidate.config["backend_version"] == "1.3.0rc10" for candidate in candidates ) + assert all( + candidate.config["forward_pass_estimators"]["agg"]["config"]["model"] + == "deepseek-ai/DeepSeek-V3" + for candidate in candidates + ) assert candidates[0].metrics["gpu_hours"] == 1.0 assert factory.worker_ids == [0] assert all(isinstance(spec, ReplaySpec) for spec in factory.runner.specs) + assert all( + spec.backend_deployment.forward_pass_estimators["agg"].backend_version + == "1.3.0rc10" + for spec in factory.runner.specs + ) assert factory.runner.closed -def test_pinned_backend_version_bypasses_latest_resolution(monkeypatch): +def test_resolved_pinned_backend_version_reaches_candidates(monkeypatch): branch = _branch(_pc()) monkeypatch.setattr( search_mod, @@ -194,8 +240,14 @@ def test_pinned_backend_version_bypasses_latest_resolution(monkeypatch): ) monkeypatch.setattr( search_mod, - "resolve_backend_version", - lambda *args: (_ for _ in ()).throw(AssertionError("must not resolve latest")), + "ForwardPassEstimatorResolver", + lambda search_space: _StaticResolver( + { + "trtllm": _forward_pass_estimator_spec( + version=search_space.requested_backend_version("trtllm") + ) + } + ), ) config = _config() config.search_space.backend_version = "0.18.0" @@ -632,6 +684,38 @@ def capabilities(self): assert factory.worker_ids == [] +def test_forward_pass_estimator_identity_fails_before_replay_execution(monkeypatch): + factory = _FakeRunnerFactory() + branch_called = False + + class FailingResolver: + def __init__(self, search_space): + del search_space + + def resolve_candidate(self, sample): + del sample + raise ValueError("pinned forward-pass estimator unavailable") + + def enumerate_branch(*args, **kwargs): + nonlocal branch_called + branch_called = True + return [_branch(_pc())] + + monkeypatch.setattr(search_mod, "ForwardPassEstimatorResolver", FailingResolver) + monkeypatch.setattr(search_mod, "enumerate_branches", enumerate_branch) + + candidates = _run_sweep( + _config(), + runner_factory=factory, + sampler_factory=_FakeSampler, + show_progress=False, + ) + + assert branch_called + assert candidates == [] + assert factory.runner.calls == 0 + + def test_unsupported_backend_pair_never_reaches_runner(monkeypatch): pc = _pc() branch = BranchSpace( @@ -815,7 +899,14 @@ def test_projection_stall_only_stops_current_branch(monkeypatch): lambda config, *, max_seq_len=None, runner_capabilities=None: [agg, disagg], ) monkeypatch.setattr( - search_mod, "resolve_backend_version", lambda hw, be: "1.3.0rc10" + search_mod, + "ForwardPassEstimatorResolver", + lambda search_space: _StaticResolver( + { + backend: _forward_pass_estimator_spec(backend) + for backend in search_space.backend + } + ), ) seen = [] diff --git a/tests/sweeper/test_search_providers.py b/tests/sweeper/test_search_providers.py index 1e6fb7121..65920f807 100644 --- a/tests/sweeper/test_search_providers.py +++ b/tests/sweeper/test_search_providers.py @@ -18,7 +18,12 @@ RuntimeHookSpec, SearchSpaceFragment, ) -from aisimulate.sweeper.replay import HookCapability, ReplayReport, RunnerCapabilities +from aisimulate.sweeper.replay import ( + ForwardPassEstimatorSpec, + HookCapability, + ReplayReport, + RunnerCapabilities, +) from aisimulate.sweeper.result import ReasonCategory from aisimulate.sweeper.sampler import Suggestion from aisimulate.sweeper.search_space import BranchSpace @@ -31,6 +36,17 @@ ) +class _StaticResolver: + def __init__(self, spec): + self.spec = spec + + def resolve_candidate(self, sample): + roles = ( + ("agg",) if sample["deployment_mode"] == "agg" else ("prefill", "decode") + ) + return {role: self.spec for role in roles} + + def _config() -> SmartSearchConfig: return SmartSearchConfig( search_space={ @@ -248,10 +264,23 @@ def _stub_branch(monkeypatch) -> None: "enumerate_branches", lambda config, *, max_seq_len=None, runner_capabilities=None: [branch], ) + forward_pass_estimator = ForwardPassEstimatorSpec( + config={ + "model": "model", + "system": "h200_sxm", + "backend": "vllm", + "backend_version": "0.11.0", + "database_mode": "SILICON", + "transfer_policy": ["xshape", "xquant", "xprofile", "xop"], + "forward_model": "op_level", + "systems_paths": ["/systems"], + }, + diagnostics={"provenance": {"selected_systems_root": "/systems"}}, + ) monkeypatch.setattr( search_module, - "resolve_backend_version", - lambda hardware, backend: "0.11.0", + "ForwardPassEstimatorResolver", + lambda search_space: _StaticResolver(forward_pass_estimator), ) @@ -376,11 +405,6 @@ def materialize_replay(self, plan, selection, context): del plan, selection, context raise InfeasibleCandidate("invalid correlated leaves") - monkeypatch.setattr( - search_module, - "resolve_backend_version", - lambda hardware, backend: "0.11.0", - ) prepared, result = search_module._materialize_one( { "deployment_mode": "agg", @@ -395,6 +419,21 @@ def materialize_replay(self, plan, selection, context): providers={"test.feature": InfeasibleAdapter()}, provider_plans={"test.feature": AdapterSearchPlan()}, runner_factory=_RunnerFactory(), + forward_pass_estimator_resolver=_StaticResolver( + ForwardPassEstimatorSpec( + config={ + "model": "model", + "system": "h200_sxm", + "backend": "vllm", + "backend_version": "0.11.0", + "database_mode": "SILICON", + "transfer_policy": ["xshape", "xquant", "xprofile", "xop"], + "forward_model": "op_level", + "systems_paths": ["/systems"], + }, + diagnostics={"provenance": {"selected_systems_root": "/systems"}}, + ) + ), ) assert prepared is None @@ -421,6 +460,7 @@ def test_runner_hook_capability_is_checked_before_runner_creation(monkeypatch) - def test_core_branch_preflight_runs_before_adapter_preparation(monkeypatch) -> None: + _stub_branch(monkeypatch) adapter = _Adapter() def reject_branches(*args, **kwargs): diff --git a/tests/sweeper/test_unified_optimizer.py b/tests/sweeper/test_unified_optimizer.py index 73fa6bfe9..3e032896b 100644 --- a/tests/sweeper/test_unified_optimizer.py +++ b/tests/sweeper/test_unified_optimizer.py @@ -13,7 +13,11 @@ ParallelShape, ReplicaParallelConfig, ) -from aisimulate.sweeper.replay import ReplayReport, RunnerCapabilities +from aisimulate.sweeper.replay import ( + ForwardPassEstimatorSpec, + ReplayReport, + RunnerCapabilities, +) from aisimulate.sweeper.sampler import ( RandomBranchSampler, SeededBayesianBranchSampler, @@ -143,6 +147,37 @@ def _branches(): ] +class _StaticResolver: + def __init__(self, specs): + self.specs = specs + + def resolve_candidate(self, sample): + roles = ( + ("agg",) if sample["deployment_mode"] == "agg" else ("prefill", "decode") + ) + return {role: self.specs[sample["backend"]] for role in roles} + + +def _forward_pass_estimator_resolver(search_space): + specs = { + backend: ForwardPassEstimatorSpec( + config={ + "model": search_space.model_name, + "system": search_space.hardware_sku, + "backend": backend, + "backend_version": "test", + "database_mode": "SILICON", + "transfer_policy": ["xshape", "xquant", "xprofile", "xop"], + "forward_model": "op_level", + "systems_paths": ["/systems"], + }, + diagnostics={"provenance": {"selected_systems_root": "/systems"}}, + ) + for backend in search_space.backend + } + return _StaticResolver(specs) + + def test_global_trial_budget_is_split_across_branches(monkeypatch) -> None: _CountingSampler.created = [] _CountingSampler.suggestion_batches = [] @@ -150,7 +185,11 @@ def test_global_trial_budget_is_split_across_branches(monkeypatch) -> None: monkeypatch.setattr( search_module, "enumerate_branches", lambda *args, **kwargs: _branches() ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "test") + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) config = SmartSearchConfig.model_validate( { "search_space": { @@ -197,7 +236,11 @@ def test_global_trial_budget_runs_branch_batches_round_robin(monkeypatch) -> Non monkeypatch.setattr( search_module, "enumerate_branches", lambda *args, **kwargs: _branches() ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "test") + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) config = SmartSearchConfig.model_validate( { "search_space": { @@ -249,7 +292,11 @@ def test_legacy_rounds_remain_branch_major_without_max_trials(monkeypatch) -> No monkeypatch.setattr( search_module, "enumerate_branches", lambda *args, **kwargs: _branches() ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "test") + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) config = SmartSearchConfig.model_validate( { "search_space": { @@ -294,7 +341,11 @@ def test_branch_seed_is_stable_when_branch_order_changes(monkeypatch) -> None: "enumerate_branches", lambda *args, **kwargs: list(reversed(_branches())), ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "test") + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) config = SmartSearchConfig.model_validate( { "search_space": { @@ -361,7 +412,11 @@ def test_candidate_timeout_applies_with_parallelism_one(monkeypatch) -> None: "enumerate_branches", lambda *args, **kwargs: [_branches()[0]], ) - monkeypatch.setattr(search_module, "resolve_backend_version", lambda *args: "test") + monkeypatch.setattr( + search_module, + "ForwardPassEstimatorResolver", + _forward_pass_estimator_resolver, + ) config = SmartSearchConfig.model_validate( { "search_space": { diff --git a/tests/test_aic.py b/tests/test_aic.py index dbbb5f530..635645320 100644 --- a/tests/test_aic.py +++ b/tests/test_aic.py @@ -49,6 +49,70 @@ def estimate(**kwargs): assert "nextn" not in calls[0] +def test_materializer_uses_the_same_canonical_config_as_replay(monkeypatch) -> None: + calls = [] + + def estimate(**kwargs): + calls.append(kwargs) + return 222 + + monkeypatch.setattr(aic, "estimate_num_gpu_blocks", estimate) + lowered = aic.materialize_aic_num_gpu_blocks( + { + "block_size": 32, + "max_num_batched_tokens": 4096, + "max_num_seqs": 19, + "free_gpu_memory_fraction": 0.8, + "timing_model": { + "type": "external", + "provider": "aic", + "config": { + "model": "test-model", + "system": "b200_sxm", + "backend": "trtllm", + "backend_version": "1.2.3", + "tp": 4, + "pp": 2, + "attention_dp": 2, + "moe_tp_size": 2, + "moe_ep_size": 4, + "gemm_quant_mode": "fp8_block", + "systems_paths": ["/resolved/systems"], + "fallback_policy": "error", + }, + }, + } + ) + + assert lowered["num_gpu_blocks"] == 222 + assert lowered["dp_size"] == 2 + assert calls == [ + { + "backend_name": "trtllm", + "system": "b200_sxm", + "model_path": "test-model", + "tp_size": 4, + "block_size": 32, + "max_num_batched_tokens": 4096, + "max_num_sequences": 19, + "gpu_memory_utilization": None, + "mem_fraction_static": None, + "free_gpu_memory_fraction": 0.8, + "backend_version": "1.2.3", + "pp_size": 2, + "moe_tp_size": 2, + "moe_ep_size": 4, + "attention_dp_size": 2, + "gemm_dtype": "fp8_block", + "moe_dtype": None, + "fmha_dtype": None, + "kv_cache_dtype": None, + "comm_dtype": None, + "systems_path": "/resolved/systems", + } + ] + + def test_capacity_wrapper_owns_backend_defaults_and_quant_normalization( monkeypatch, ) -> None: diff --git a/tests/test_cli_config.py b/tests/test_cli_config.py index 825e4a8bf..eb2bfc333 100644 --- a/tests/test_cli_config.py +++ b/tests/test_cli_config.py @@ -287,6 +287,7 @@ def test_engine_scheduler_domains_replace_defaults_and_preserve_log_scale() -> N "engine": { **_engine(), "mode": "aggregated", + "backend": "vllm", "backend_version": "0.19.0", "context_length": 4096, "workers": {