"""Generate the release report from raw evidence, not hand-entered benchmark numbers.""" import json import statistics import sys from pathlib import Path def report(diagnostic_path, stress_path): diagnostic = json.loads(Path(diagnostic_path).read_text()) stress = json.loads(Path(stress_path).read_text()) lines = [ "# PCD measurements and limitations", "", "Measured 2026-09-16 on one NVIDIA L40S, FP16, PyTorch 2.14.0, Transformers 5.17.0, SDPA attention and reference PyTorch convolution. No optimized causal-conv1d, torch.compile, vLLM/SGLang or Jev service comparison.", "", "## Release decision", "", "**Experimental inference-only release, not a production-quality or calibrated Jev substitute.** The latency goal is met on multi-field examples. The proposed accuracy non-inferiority goal is **not met**: use application-specific evaluation, an autoregressive fallback, or separately funded training before relying on it. No always-on service was deployed.", "", "## Development and fresh audit", "", "The 12 development cases were used to debug/select prompts. Six additional cases were written after freezing the protocol and not used to retune it. Both sets are tiny, hand-authored, and not representative production data. Each method/case has one untimed warmup and three measured repetitions. Accuracy counts unique cases, not repeats.", "", "| Split | Method | Mean ms | Median ms | p95 ms | Field accuracy | Exact objects | Schema valid |", "|---|---|---:|---:|---:|---:|---:|---:|", ] for split in ("development", "audit"): for method in ("token", "sequence", "ar_sequence", "ar_token"): s = diagnostic["split_summary"][split][method] lines.append( f"| {split} | {method} | {s['mean_ms']:.2f} | {s['median_ms']:.2f} | {s['p95_ms']:.2f} | {s['field_accuracy']:.1%} | {s['exact_accuracy']:.1%} | {s['schema_compliance']:.1%} |" ) lines += [ "", "`ar_sequence` is greedy direct-answer JSON using original values and the same sequence-mode system/schema. `ar_token` uses the code catalog and an all-field JSON request; codes are strictly decoded, not repaired. Token PCD asks independent per-field questions, so its suffix prompt and distribution are not identical to joint AR. All use the same model, precision and backend. Empty thinking is explicit for these direct-answer modes.", "", "## Synthetic scaling (one distinct input per row)", "", "| Case | Token PCD ms | Sequence PCD ms | AR JSON ms | AR/token ratio | Token / sequence / AR field accuracy |", "|---|---:|---:|---:|---:|---|", ] for case in stress["cases"]: groups = { method: [ r for r in stress["rows"] if r["case_id"] == case["id"] and r["method"] == method ] for method in ("token", "sequence", "ar_sequence") } means = { method: statistics.mean(r["result"]["elapsed_ms"] for r in rows) for method, rows in groups.items() } accuracies = [ f"{rows[0]['evaluation']['field_correct'] / rows[0]['evaluation']['field_total']:.0%}" for rows in groups.values() ] lines.append( f"| {case['id']} | {means['token']:.2f} | {means['sequence']:.2f} | {means['ar_sequence']:.2f} | {means['ar_sequence'] / means['token']:.1f}x | {' / '.join(accuracies)} |" ) lines += [ "", "**High-cardinality warning:** token mode chose the wrong answer in both the 64- and 255-category probes. Sequence mode was correct on these inputs, but the 255-category sequence scorer was slower than AR. Do not interpret the high aggregate boolean accuracy as high-cardinality accuracy. The 28-field example is literal configuration extraction, not a broad semantic understanding benchmark.", "", "## Incorrect constrained outputs (one per unique case)", "", "| Case | Method | Expected | Actual |", "|---|---|---|---|", ] expected = {case["id"]: case["expected"] for case in diagnostic["cases"] + stress["cases"]} for row in diagnostic["rows"] + stress["rows"]: if ( row["repeat"] or row["method"] not in {"token", "sequence"} or row["evaluation"]["exact_match"] ): continue gold = json.dumps(expected[row["case_id"]], ensure_ascii=False).replace("|", "\\|") actual = row["result"]["text"].replace("|", "\\|").replace("\n", " ") lines.append(f"| {row['case_id']} | {row['method']} | `{gold}` | `{actual}` |") lines += ["", "## Correctness and resource evidence", ""] for name, data in [("diagnostic", diagnostic), ("stress", stress)]: validation = data["validation"] errors = {r["mode"]: r["max_score_error"] for r in validation["score_comparisons"]} peak = max(r["peak_allocated_bytes"] for r in data["rows"]) / 2**30 lines.append( f"- {name}: validation passed; maximum candidate-score differences against uncached reference {errors}; peak allocated VRAM {peak:.2f} GiB; in-function time {data['wall_seconds']:.2f}s." ) lines += [ "- Local tests exercise a tiny FP32 hybrid model; CUDA validation also tests the actual model in FP32 and FP16. Both single-token and multi-token convolution cache paths preserve the original prefix state. Score tolerances are 0.002 in FP32 and 0.10 in FP16; raw measured errors are retained, not assumed zero.", "- An early BF16 hidden-state check failed (maximum observed difference 0.25 with 0.15 absolute tolerance). We did not relax that BF16 gate; FP16 is the recommended tested runtime precision. Original stored BF16 weights remain unchanged.", "- Candidate probabilities are explicitly uncalibrated. No ECE, Brier-score calibration, learned confidence threshold, or proprietary Jev RLCD training is claimed.", "- Request timings include prompt tokenization, cached-schema lookup, prefill, cache forking, all microbatches, candidate scoring, and serialization; CUDA is synchronized. They exclude model loading, network, provisioning and output evaluation. Initial schema compilation and cold starts are not benchmarked separately.", "- p95 estimates are descriptive with very small samples. GPU clocks/allocator/runtime effects varied between exploratory jobs; the report uses the final fixed-source runs, not the fastest observed earlier latency.", "- Native reasoning was sampled during development on three cases with a 256-token cap: two truncated and one emitted fenced JSON. Those are not valid completed baseline results and are not used to claim PCD speedups or native-model incompetence. We avoided paying to repeat that capped experiment.", "- Frugal execution: serial single-L40S jobs, cached public weights, bounded runtimes, no H100, no training and no permanent deployment. In-function seconds are not a billing statement; startup/idle/CPU/storage charges are additional.", "", "## Raw release evidence", "", f"- [Diagnostic/audit JSON](../{diagnostic_path})", f"- [Scaling JSON](../{stress_path})", "", "Raw files include every measured output, score, token/branch count, version and inference-source hash. The selected prompt was not changed after the fresh audit. Earlier iterations are development artifacts and are not release performance claims.", ] target = Path("docs/pcd-results.md") target.write_text("\n".join(lines) + "\n") print(target) if __name__ == "__main__": report(*sys.argv[1:])