Spaces:
Running
Running
File size: 14,199 Bytes
0c6c82c 44745f2 0c6c82c 44745f2 ce2d64b 44745f2 e5c4ee4 d2258e5 94910ac 8f91935 ce2d64b e5c4ee4 44745f2 ce2d64b 0c6c82c 8f91935 ce2d64b 0c6c82c 20fb354 0c6c82c 20fb354 0c6c82c 20fb354 44745f2 ce2d64b 44745f2 8f91935 ce2d64b 44745f2 0c6c82c 20fb354 44745f2 ce2d64b 44745f2 0c6c82c ce2d64b e5c4ee4 94910ac d2258e5 e5c4ee4 8f91935 ce2d64b 0c6c82c 44745f2 ce2d64b 44745f2 ce2d64b e5c4ee4 94910ac d2258e5 8f91935 ce2d64b 0c6c82c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 | from __future__ import annotations
import importlib
import re
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
README = (ROOT / "README.md").read_text()
SRC = ROOT / "src"
sys.path.insert(0, str(SRC))
inferscale = importlib.import_module("inferscale")
internal_version = inferscale.__version__
design_space_search = inferscale.design_space_search
compare_agent_policies = inferscale.compare_agent_policies
adaptive_tiering_study = inferscale.adaptive_tiering_study
adaptive_alpha_sweep = inferscale.adaptive_alpha_sweep
compare_agent_memory_policies = inferscale.compare_agent_memory_policies
agent_memory_budget_sweep = inferscale.agent_memory_budget_sweep
agent_affinity_sweep = inferscale.agent_affinity_sweep
execution_prefetch_study = inferscale.execution_prefetch_study
execution_threshold_sweep = inferscale.execution_threshold_sweep
execution_decay_sweep = inferscale.execution_decay_sweep
run_execution_learning = inferscale.run_execution_learning
paired_study = inferscale.paired_study
robustness_study = inferscale.robustness_study
run_agent_session_simulation = inferscale.run_agent_session_simulation
ttl_retention_sweep = inferscale.ttl_retention_sweep
run_simulation = inferscale.run_simulation
validate_cases = inferscale.validate_cases
errors: list[str] = []
match = re.search(r"^short_description:\s*(.+)$", README, re.MULTILINE)
if not match:
errors.append("README metadata is missing short_description")
short = ""
else:
short = match.group(1).strip().strip('"\'')
if len(short) > 60:
errors.append(f"short_description is {len(short)} chars; HF limit is 60")
if "sdk: static" not in README:
errors.append("README metadata must use sdk: static")
if internal_version != "0.8.0":
errors.append(f"internal package version is {internal_version}; expected 0.8.0")
# Public-facing release/version branding is intentionally absent. Model names
# such as Mistral-7B-v0.3 are allowed; project headings/badges are not.
public_texts = [("README.md", README), ("index.html", (ROOT / "index.html").read_text())]
for name, text in public_texts:
if re.search(r"InferScale(?:-Sim)?\s*/?\s*v\d", text, re.IGNORECASE):
errors.append(f"{name} contains public project version branding")
if (ROOT / "CHANGELOG.md").exists():
errors.append("CHANGELOG.md should be omitted from the public portfolio release")
src_files = sorted((ROOT / "src" / "inferscale").glob("*.py"))
web_files = sorted((ROOT / "py" / "inferscale").glob("*.py"))
if [path.name for path in src_files] != [path.name for path in web_files]:
errors.append("browser Python mirror is stale; run python scripts/sync_web_python.py")
else:
for src, web in zip(src_files, web_files, strict=True):
if src.read_bytes() != web.read_bytes():
errors.append(f"browser mirror differs for {src.name}; run sync_web_python.py")
worker_text = (ROOT / "worker.mjs").read_text()
for src in src_files:
if f'"{src.name}"' not in worker_text:
errors.append(f"worker module list is missing {src.name}")
for ui_file in (ROOT / "index.html", ROOT / "app.js"):
try:
ui_file.read_text().encode("ascii")
except UnicodeEncodeError:
errors.append(f"{ui_file.name} contains non-ASCII UI glyphs; use text labels for reliable rendering")
index = (ROOT / "index.html").read_text()
app = (ROOT / "app.js").read_text()
if "<footer" in index.lower():
errors.append("UI should not include a product-style footer")
if "Download PNG" not in index or ".chart-download" not in app:
errors.append("chart PNG export controls are missing")
if "Worst repetition" not in index or "Target" not in index:
errors.append("capacity evidence columns are missing")
for expected in ["Trace replay", "Research Studies", "Run paired study", "Stress-test selected hypothesis", "Agent Sessions", "Session Policy Arena", "TTL frontier", "Agent Memory Lab", "Affinity Frontier", "Stress HBM budget", "Predictive Tiering Lab", "Compare predictive policies", "Sweep adaptation rate", "Execution Learning", "Prefetch Policy Study", "Confidence Threshold Study", "Forgetting-Rate Study", "Download JSON"]:
if expected not in index:
errors.append(f"UI is missing research/trace feature: {expected}")
# Every $("id") lookup in app.js should resolve to a static DOM id.
app_ids = set(re.findall(r'\$\("([A-Za-z0-9_-]+)"\)', app))
html_ids = set(re.findall(r'id="([A-Za-z0-9_-]+)"', index))
missing_ids = sorted(app_ids - html_ids)
if missing_ids:
errors.append(f"app.js references missing DOM ids: {', '.join(missing_ids[:12])}")
smoke_cfg = {
"model": "Qwen2.5-3B",
"accelerator": "L4",
"quantization": "int8",
"duration_s": 4,
"request_rate_rps": 1,
"prompt_tokens_mean": 128,
"output_tokens_mean": 8,
}
try:
smoke = run_simulation(smoke_cfg)
if smoke["summary"]["requests_completed"] <= 0:
errors.append("simulation smoke test completed zero requests")
if smoke["provenance"]["latency_profile_type"] != "analytical-reference":
errors.append("profile provenance guard is missing")
if smoke["diagnostics"].get("provenance") != "heuristic-simulator-diagnosis":
errors.append("diagnosis provenance missing")
except Exception as exc: # pragma: no cover
errors.append(f"colocated smoke test raised: {exc}")
try:
trace = run_simulation(smoke_cfg | {
"arrival_process": "trace",
"trace_requests": [
{"arrival_time": 0.0, "prompt_tokens": 64, "output_tokens": 4},
{"arrival_time": 0.2, "prompt_tokens": 96, "output_tokens": 6},
],
})
if trace["summary"]["requests_generated"] != 2:
errors.append("trace replay smoke test did not preserve request count")
except Exception as exc: # pragma: no cover
errors.append(f"trace replay smoke test raised: {exc}")
try:
pd = run_simulation(smoke_cfg | {
"topology": "disaggregated_pd",
"scheduler": "continuous_slo",
"prefill_accelerator": "L4",
"decode_accelerator": "L4",
"interconnect_gbps": 50,
})
if pd["resource"].get("p95_transfer_ms", 0) <= 0:
errors.append("P/D transfer telemetry missing")
except Exception as exc: # pragma: no cover
errors.append(f"P/D smoke test raised: {exc}")
try:
design = design_space_search(smoke_cfg | {
"shared_prefix_tokens": 64,
"prefix_reuse_fraction": 0.5,
"prefill_accelerator": "L4",
"decode_accelerator": "L4",
}, include_disaggregated=False)
if design["candidate_count"] != 10 or design["pareto_count"] < 1 or design["efficiency_pareto_count"] < 1:
errors.append("design-space smoke test did not return both expected Pareto frontiers")
except Exception as exc: # pragma: no cover
errors.append(f"design-space smoke test raised: {exc}")
try:
paired = paired_study(
smoke_cfg | {"shared_prefix_tokens": 64, "prefix_reuse_fraction": 0.75},
study="prefix_cache",
repetitions=4,
bootstrap_samples=100,
)
if paired["protocol"] != "paired-common-random-numbers" or len(paired["metrics"]) != 4:
errors.append("paired research study smoke test is incomplete")
except Exception as exc: # pragma: no cover
errors.append(f"paired study smoke test raised: {exc}")
try:
robust = robustness_study(
smoke_cfg | {"prefill_accelerator": "L4", "decode_accelerator": "L4"},
study="pd_vs_colocated",
samples=4,
uncertainty=0.10,
)
if robust["method"] != "shared-multiplicative-latency-perturbation" or len(robust["rows"]) != 4:
errors.append("robustness study smoke test is incomplete")
except Exception as exc: # pragma: no cover
errors.append(f"robustness study smoke test raised: {exc}")
try:
agent_cfg = {
"model": "Qwen2.5-3B", "accelerator": "L4", "quantization": "int8",
"duration_s": 12, "session_rate_rps": 0.25, "replicas": 2, "seed": 7,
"retention_policy": "ttl", "routing_policy": "session_affinity", "kv_ttl_s": 3,
}
agent = run_agent_session_simulation(agent_cfg)
if agent["provenance"].get("mode") != "stateful-agent-session-simulation":
errors.append("agent-session provenance guard is missing")
if agent["summary"].get("turns_completed", 0) <= 0:
errors.append("agent-session smoke test completed zero turns")
agent_compare = compare_agent_policies(agent_cfg)
if agent_compare.get("candidate_count") != 4:
errors.append("agent policy arena did not return four candidates")
agent_ttl = ttl_retention_sweep(agent_cfg, [0, 1, 3])
if len(agent_ttl.get("rows", [])) != 3 or agent_ttl.get("pareto_count", 0) < 1:
errors.append("agent TTL frontier smoke test is incomplete")
agent_memory = compare_agent_memory_policies(agent_cfg | {"host_memory_gb": 4})
if agent_memory.get("candidate_count") != 5 or not any(row.get("host_hit_rate", 0) > 0 for row in agent_memory.get("rows", [])):
errors.append("agent tiered-memory comparison is incomplete")
agent_budget = agent_memory_budget_sweep(agent_cfg | {"host_memory_gb": 4}, [0.5, 1.0])
if len(agent_budget.get("rows", [])) != 6:
errors.append("finite HBM budget study is incomplete")
agent_affinity = agent_affinity_sweep(agent_cfg, [0, 150, 600])
if len(agent_affinity.get("rows", [])) != 3:
errors.append("bounded-affinity sweep is incomplete")
adaptive = run_agent_session_simulation(agent_cfg | {
"retention_policy": "adaptive",
"routing_policy": "bounded_affinity",
"adaptive_predictor_scope": "per_tool_ema",
})
if adaptive["provenance"].get("adaptive_policy") != "online-tool-gap-ewma-no-lookahead":
errors.append("adaptive policy provenance guard is missing")
if adaptive["resource"].get("adaptive_prediction_count", 0) <= 0:
errors.append("adaptive policy produced no tool-gap predictions")
predictive = adaptive_tiering_study(
agent_cfg | {"host_memory_gb": 4, "routing_policy": "bounded_affinity"},
horizon_s=40, shift_fraction=0.5, shift_multiplier=2.0, alpha=0.3,
)
if len(predictive.get("rows", [])) != 5 or predictive.get("shift_observation", 0) <= 0:
errors.append("predictive tiering study is incomplete")
alpha_sweep = adaptive_alpha_sweep(
agent_cfg | {"host_memory_gb": 4, "routing_policy": "bounded_affinity"},
[0.1, 0.3, 0.8], horizon_s=40, shift_fraction=0.5, shift_multiplier=2.0,
)
if len(alpha_sweep.get("rows", [])) != 3:
errors.append("adaptation-rate sweep is incomplete")
except Exception as exc: # pragma: no cover
errors.append(f"agent-session smoke test raised: {exc}")
try:
execution_cfg = {
"model": "Qwen2.5-3B", "accelerator": "L4", "quantization": "int8",
"duration_s": 50, "workflow_rate_rps": 0.2, "seed": 7,
"shift_fraction": 0.5, "confidence_threshold": 0.5,
}
execution_run = run_execution_learning(execution_cfg | {"prefetch_policy": "decayed"})
if execution_run["provenance"].get("mode") != "online-agent-execution-learning":
errors.append("execution-learning provenance guard is missing")
if execution_run["prediction"].get("count", 0) <= 0:
errors.append("execution-learning run produced no transition observations")
execution_compare = execution_prefetch_study(execution_cfg)
if len(execution_compare.get("rows", [])) != 4:
errors.append("execution prefetch policy study is incomplete")
execution_threshold = execution_threshold_sweep(execution_cfg, [0.0, 0.5, 0.9])
if len(execution_threshold.get("rows", [])) != 3:
errors.append("execution confidence-threshold sweep is incomplete")
execution_decay = execution_decay_sweep(execution_cfg, [0.5, 0.85, 1.0])
if len(execution_decay.get("rows", [])) != 3:
errors.append("execution transition-decay sweep is incomplete")
except Exception as exc: # pragma: no cover
errors.append(f"execution-learning smoke test raised: {exc}")
try:
validation = validate_cases([{
"name": "release-fixture",
"config": smoke_cfg,
"measured": {"p95_ttft_ms": 100.0, "goodput_rps": 0.8},
}])
if validation["observation_count"] != 2:
errors.append("external-measurement validation hook failed")
except Exception as exc: # pragma: no cover
errors.append(f"validation hook smoke test raised: {exc}")
if errors:
print("InferScale release check: FAIL")
for error in errors:
print(f"- {error}")
raise SystemExit(1)
print("InferScale release check: PASS")
print(f"HF short_description: {len(short)}/60 characters")
print(f"Python modules mirrored: {len(src_files)}")
print(f"Colocated smoke requests: {smoke['summary']['requests_completed']}")
print(f"Trace replay requests: {trace['summary']['requests_generated']}")
print(f"P/D transfer p95: {pd['resource']['p95_transfer_ms']:.3f} ms")
print(f"Design candidates: {design['candidate_count']}")
print(f"Paired-study metrics: {len(paired['metrics'])}")
print(f"Robustness perturbations: {len(robust['rows'])}")
print(f"Agent turns: {agent['summary']['turns_completed']}")
print(f"Agent policy candidates: {agent_compare['candidate_count']}")
print(f"Agent TTL candidates: {len(agent_ttl['rows'])}")
print(f"Agent memory policies: {agent_memory['candidate_count']}")
print(f"HBM budget study rows: {len(agent_budget['rows'])}")
print(f"Affinity sweep points: {len(agent_affinity['rows'])}")
print(f"Adaptive predictions: {adaptive['resource']['adaptive_prediction_count']}")
print(f"Predictive-tiering candidates: {len(predictive['rows'])}")
print(f"Adaptation-rate points: {len(alpha_sweep['rows'])}")
print(f"Execution transition observations: {execution_run['prediction']['count']}")
print(f"Execution policy candidates: {len(execution_compare['rows'])}")
print(f"Execution threshold points: {len(execution_threshold['rows'])}")
print(f"Execution decay points: {len(execution_decay['rows'])}")
print(f"Validation observations: {validation['observation_count']}")
print(f"Profile provenance: {smoke['provenance']['latency_profile_type']}")
|