Xunzhuo commited on
Commit
23fdf61
·
1 Parent(s): 8901c75

Publish clean Decision model repository

Browse files

Signed-off-by: Xunzhuo <[email protected]>

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. Dockerfile.runtime +0 -12
  2. MATERIALS.json +0 -28
  3. NORMALIZATION_RUNTIME.md +0 -5
  4. NULL_DESCRIPTION_RENDERING.json +0 -7
  5. README.md +14 -52
  6. RUNTIME-RELEASE.json +0 -11
  7. RUNTIME.md +0 -52
  8. RUNTIME_BINDING.json +0 -76
  9. SERVING_OPTIMIZATION.json +0 -34
  10. SOURCE_BUNDLE_MANIFEST.json +0 -173
  11. USAGE.md +0 -59
  12. WEIGHTING.md +0 -21
  13. assets/decision-expanded-old_core-600px.png +0 -3
  14. assets/decision-expanded-old_core.pdf +0 -0
  15. assets/decision-expanded-old_core.png +0 -3
  16. assets/decision-expanded-old_core.svg +0 -1620
  17. assets/decision-expanded-overview-600px.png +0 -0
  18. assets/decision-expanded-overview.pdf +0 -0
  19. assets/decision-expanded-overview.png +0 -3
  20. assets/decision-expanded-overview.svg +0 -804
  21. assets/decision-expanded-ranking-600px.png +0 -0
  22. assets/decision-expanded-ranking.pdf +0 -0
  23. assets/decision-expanded-ranking.png +0 -3
  24. assets/decision-expanded-ranking.svg +0 -501
  25. assets/decision-expanded-v3_core-600px.png +0 -3
  26. assets/decision-expanded-v3_core.pdf +0 -0
  27. assets/decision-expanded-v3_core.png +0 -3
  28. assets/decision-expanded-v3_core.svg +0 -1621
  29. assets/decision-expanded-v4-600px.png +0 -0
  30. assets/decision-expanded-v4.pdf +0 -0
  31. assets/decision-expanded-v4.png +0 -3
  32. assets/decision-expanded-v4.svg +0 -668
  33. assets/decision-expanded-v5-600px.png +0 -0
  34. assets/decision-expanded-v5.pdf +0 -0
  35. assets/decision-expanded-v5.png +0 -3
  36. assets/decision-expanded-v5.svg +0 -804
  37. assets/decision-family-header.png +0 -3
  38. assets/decision-matrix.pdf +0 -0
  39. assets/decision-matrix.svg +0 -1338
  40. assets/decision-question-scaling-600px.png +0 -0
  41. assets/decision-question-scaling.pdf +0 -0
  42. assets/decision-question-scaling.svg +0 -239
  43. assets/decision-ranking.pdf +0 -0
  44. assets/decision-ranking.svg +0 -369
  45. bundle-manifest.json +0 -243
  46. code/decision_api.py +0 -198
  47. code/decision_model.py +0 -173
  48. code/predict.py +0 -43
  49. code/profile_guard.py +0 -82
  50. code/runtime_profile.py +0 -36
Dockerfile.runtime DELETED
@@ -1,12 +0,0 @@
1
- # Public base verified by manifest digest and critical PyTorch file hashes.
2
- # CPU build/import validation is separate from model GPU qualification; see RUNTIME.md.
3
- FROM vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339
4
- COPY runtime-fla-requirements.lock /tmp/runtime-fla-requirements.lock
5
- RUN python3 -m pip install --no-cache-dir --no-index --no-deps --require-hashes --target /opt/decision-fla -r /tmp/runtime-fla-requirements.lock
6
- ENV PYTHONPATH=/opt/decision-fla
7
- COPY pyproject.toml /opt/decision-wrapper/pyproject.toml
8
- COPY src/ /opt/decision-wrapper/src/
9
- RUN python3 -m pip install --no-cache-dir --no-deps --no-build-isolation /opt/decision-wrapper
10
- WORKDIR /model
11
- ENTRYPOINT []
12
- CMD ["/bin/bash"]
 
 
 
 
 
 
 
 
 
 
 
 
 
MATERIALS.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "kind": "latest-public-comparison-documents",
3
- "public_models": 15,
4
- "scored_questions": 3766,
5
- "tasks": 54,
6
- "source_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
7
- "files": {
8
- ".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
9
- "DIAGNOSTICS.md": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa",
10
- "EVALUATION.md": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a",
11
- "QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
12
- "README.md": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
13
- "SENSITIVITY.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
14
- "TASKS.md": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247",
15
- "USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
16
- "WEIGHTING.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
17
- "assets/decision-matrix.pdf": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b",
18
- "assets/decision-matrix.png": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696",
19
- "assets/decision-matrix.svg": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc",
20
- "assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
21
- "assets/decision-ranking.pdf": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f",
22
- "assets/decision-ranking.png": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9",
23
- "assets/decision-ranking.svg": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e",
24
- "metrics/benchmark.json": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c",
25
- "metrics/evaluation-provenance.json": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9",
26
- "metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
27
- }
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
NORMALIZATION_RUNTIME.md DELETED
@@ -1,5 +0,0 @@
1
- # Validated normalization runtime
2
-
3
- This Nox bundle installs its recorded FLA normalization profile through the default public entrypoint. It requires the pinned ROCm runtime on gfx942 and a fresh process. The profile covers dimension 128, BF16 input/output, FP32 reciprocal norms, and buckets 1–64 for the actual 32 normalized value heads, batch size at most 8 and complete inputs at most 16,384 tokens. Unknown keys fail closed. Weights, tokenizer, prompt, readout and temperature remain unchanged from the source export.
4
-
5
- No private compilation cache is required. Other FLA kernels retain normal runtime behavior. Numerical validation of this newly bound bundle is still required. Earlier latency measurements do not measure this runtime or candidate.
 
 
 
 
 
 
NULL_DESCRIPTION_RENDERING.json DELETED
@@ -1,7 +0,0 @@
1
- {
2
- "rule": "For Choice only, replace a null description with its original key text before tokenization.",
3
- "preserved": "Non-null descriptions including empty string; keys/order; Noul and Score; original request; weights/tokenizer/temperature/numeric runtime.",
4
- "source_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
5
- "qualification": "Candidate only; full fixed benchmark and independent public loading proof required.",
6
- "not_a_weight_training_update": true
7
- }
 
 
 
 
 
 
 
 
README.md CHANGED
@@ -7,7 +7,6 @@ tags:
7
  - decision-model
8
  - classification
9
  - qwen3_5
10
- - custom-code
11
  - pytorch
12
  - rocm
13
  ---
@@ -58,75 +57,38 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading *
58
 
59
  ![Capability matrix](assets/decision-matrix.png)
60
 
61
- [All 54 tasks](TASKS.md) · [Order, missing-evidence and calibration diagnostics](DIAGNOSTICS.md) · [Methods and uncertainty](EVALUATION.md)
62
 
63
  ## More questions, one request
64
 
65
  ![Question-count latency](assets/decision-question-scaling.png)
66
 
67
- Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](QUESTION-SCALING.md).
68
 
69
- ## Use Nox-4B
70
-
71
- Use the [official TypeSafe Python SDK](https://docs.typesafe.ai/sdk/python/usage) with your SystemOne-compatible endpoint, configured to serve `Decision-1.0-Nox-4B`. Replace the example URL and API key with your own.
72
 
73
  ```bash
74
- pip install typesafe-sdk
75
  ```
76
 
77
- ```python
78
- from typesafe_sdk import Choice, Noul, TypeSafeClient
79
-
80
- with TypeSafeClient(
81
- api_key="YOUR_ENDPOINT_API_KEY",
82
- base_url="https://your-decision-endpoint.example",
83
- model="Decision-1.0-Nox-4B",
84
- ) as client:
85
- result = client.system_one(
86
- state="Customer reports a duplicate charge and asks for a refund.",
87
- questions={
88
- "route": Choice(
89
- instructions="Which team should handle this request?",
90
- criteria={"billing": "Payments and refunds", "technical": "Product faults"},
91
- ),
92
- "refund_requested": Noul(instructions="Did the customer request a refund?"),
93
- },
94
- )
95
- print(result.choices["route"].choice)
96
- print(result.nouls["refund_requested"].noul)
97
- ```
98
 
99
- The same request with curl:
 
 
100
 
101
  ```bash
102
  curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
103
  -H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
104
  -H 'Content-Type: application/json' \
105
- --data-raw '{
106
- "model": "Decision-1.0-Nox-4B",
107
- "state": "Customer reports a duplicate charge and asks for a refund.",
108
- "questions": {
109
- "route": {
110
- "type": "choice",
111
- "instructions": "Which team should handle this request?",
112
- "criteria": {
113
- "billing": "Payments and refunds",
114
- "technical": "Product faults"
115
- }
116
- },
117
- "refund_requested": {
118
- "type": "noul",
119
- "instructions": "Did the customer request a refund?"
120
- }
121
- }
122
- }'
123
  ```
124
 
125
- [Typed request and response guide](USAGE.md) · [Model runtime requirements](RUNTIME.md)
126
-
127
- Choice candidates with a null description use their ID text, which may increase input tokens.
128
 
129
- The complete state, question and candidates must fit 16,384 tokens; overflow is rejected. The bundled normalization profile loads automatically. AMD gfx942 is validated; CPU/MPS are unsupported and NVIDIA is unqualified. Use a fresh Python process when switching profiles.
130
 
131
  ## Architecture
132
 
@@ -134,6 +96,6 @@ The complete state, question and candidates must fit 16,384 tokens; overflow is
134
 
135
  A causal Qwen3.5 text backbone combines gated linear and full attention. A shared candidate head reads candidate endpoints and the final query vector. Each question uses one forward pass; questions run independently in batches of eight.
136
 
137
- [Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg) · [Inference code](code/decision_model.py)
138
 
139
  Adapted from [Qwen3.5-4B](https://huggingface.co/Qwen/Qwen3.5-4B). It evaluates supplied evidence without live retrieval; confidence does not guarantee correctness. [License](LICENSE) · [Attributions](ATTRIBUTIONS.md).
 
7
  - decision-model
8
  - classification
9
  - qwen3_5
 
10
  - pytorch
11
  - rocm
12
  ---
 
57
 
58
  ![Capability matrix](assets/decision-matrix.png)
59
 
60
+ [All 54 tasks](evaluation/TASKS.md) · [Order, missing-evidence and calibration diagnostics](evaluation/DIAGNOSTICS.md) · [Methods and uncertainty](evaluation/EVALUATION.md)
61
 
62
  ## More questions, one request
63
 
64
  ![Question-count latency](assets/decision-question-scaling.png)
65
 
66
+ Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](evaluation/QUESTION-SCALING.md).
67
 
68
+ ## Download the complete model repository
 
 
69
 
70
  ```bash
71
+ hf download llm-semantic-router/Decision-1.0-Nox-4B --local-dir Decision-1.0-Nox-4B
72
  ```
73
 
74
+ This downloads the complete model release. The root `config.json` lists the backbone, tokenizer, decision head, and calibration files.
75
+
76
+ ## Serve with vLLM Semantic Router
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
 
78
+ This repository contains model data only. Use the vLLM Semantic Router Decision runtime to load `llm-semantic-router/Decision-1.0-Nox-4B` and serve Choice, Noul, and Score requests. The serving implementation and its dependencies live in vLLM Semantic Router; this release does not bundle executable model code. `transformers.AutoModel.from_pretrained` cannot load the custom Decision head directly.
79
+
80
+ After configuring a compatible Decision endpoint, send a [SystemOne request](https://docs.typesafe.ai/api) (replace the placeholder URL and key):
81
 
82
  ```bash
83
  curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
84
  -H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
85
  -H 'Content-Type: application/json' \
86
+ --data-raw '{"model":"Decision-1.0-Nox-4B","state":"Customer requests a refund.","questions":{"route":{"type":"choice","instructions":"Which team should handle this?","criteria":{"billing":"Payments and refunds","technical":"Product faults"}}}}'
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
87
  ```
88
 
89
+ The Hugging Face repository is a model download, not a hosted inference endpoint.
 
 
90
 
91
+ The published model's complete state, question, and candidates have a 16,384-token input limit. See the [evaluation scope](evaluation/EVALUATION.md) for measured conditions.
92
 
93
  ## Architecture
94
 
 
96
 
97
  A causal Qwen3.5 text backbone combines gated linear and full attention. A shared candidate head reads candidate endpoints and the final query vector. Each question uses one forward pass; questions run independently in batches of eight.
98
 
99
+ [Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg)
100
 
101
  Adapted from [Qwen3.5-4B](https://huggingface.co/Qwen/Qwen3.5-4B). It evaluates supplied evidence without live retrieval; confidence does not guarantee correctness. [License](LICENSE) · [Attributions](ATTRIBUTIONS.md).
RUNTIME-RELEASE.json DELETED
@@ -1,11 +0,0 @@
1
- {
2
- "release": "v1.3.2",
3
- "kind": "SystemOne Choice null-description semantics",
4
- "rule": "A null Choice description uses the original candidate key as its description.",
5
- "weights_tokenizer_temperature_unchanged": true,
6
- "explicit_descriptions_and_Noul_Score_unchanged": true,
7
- "bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
8
- "qualified_statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
9
- "exact_public_contract_proof_sha256": "ebf27c7aa2bb4a12ba2cbb27e4096e46cb43ac4a20b9d3e1bc34f62777a1e63a",
10
- "runtime_note": "This is an API rendering improvement, not a newly trained checkpoint."
11
- }
 
 
 
 
 
 
 
 
 
 
 
 
RUNTIME.md DELETED
@@ -1,52 +0,0 @@
1
- # Public ROCm runtime
2
-
3
- The package has a public, digest-pinned installation path. `Dockerfile.runtime` starts from `vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339`, adds two hash-checked FLA wheels, and installs this repository's loading wrapper. It keeps the base image's ROCm PyTorch and Triton builds. The vLLM server is not used by Decision inference.
4
-
5
- **Validation boundary:** the public registry manifest, base-image ancestry, package metadata and critical PyTorch binary hashes have been checked. The recipe built successfully and passed CPU imports and real AMD ROCm gfx942 GPU GPU inference for both released bundles. On the packaged three-question Choice/Noul/Score example, its complete responses matched the qualified research runtime exactly; the wrapper matched the direct engine and rejected an oversized complete input. Evidence is in `runtime-build-provenance.json`. This example establishes a working public installation path; it is not a full rerun of the quality or timing benchmark. Published benchmark results use the qualified runtime in each bundle's `runtime.json`.
6
-
7
- ## Build and run
8
-
9
- Use an AMD ROCm-compatible Linux host with Docker and the required GPU driver. Check the [AMD PyTorch installation and host prerequisites](https://rocm.docs.amd.com/projects/ai-ecosystem/en/latest/frameworks/pytorch/install.html). CPU and MPS inference are not supported by this model engine; no NVIDIA validation is claimed.
10
-
11
- Run these commands from the downloaded model repository containing `Dockerfile.runtime`, `runtime-fla-requirements.lock`, `pyproject.toml`, `src/`, and the model bundle:
12
-
13
- ```bash
14
- docker build --pull -f Dockerfile.runtime -t decision-runtime:1.0 .
15
- mkdir -p runtime-output
16
- docker run --rm \
17
- --device=/dev/kfd --device=/dev/dri --group-add video --ipc=host \
18
- -v "$PWD":/model:ro -v "$PWD/runtime-output":/output \
19
- decision-runtime:1.0 \
20
- python3 -m decision.example /model --local-files-only --output /output/example.json
21
- ```
22
-
23
- This loads locally without a Hub token. The example tests Choice, Noul and Score through the wrapper and compares them with the frozen direct engine; inspect its actual result rather than assuming a predicted answer. Keep the runtime checks enabled. If they report a mismatch, resolve the cause before using this environment to reproduce benchmark claims. The explicit `allow_unvalidated_runtime=True` option is for separately labeled experiments, not benchmark reproduction.
24
-
25
- The image is large because its public base includes the vLLM development environment. No Decision weights, dataset, user credentials or private runtime image are required to build it. Model weights are mounted at execution time.
26
-
27
- ## What is pinned
28
-
29
- The public base locks the existing OS, ROCm libraries and Python dependency environment by content digest. Its exact observed core versions are:
30
-
31
- | Component | Observed value |
32
- |---|---|
33
- | Python | 3.12.13 |
34
- | PyTorch | 2.12.0+git6bbd260 |
35
- | PyTorch commit | 6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5 |
36
- | ROCm userspace / HIP build | 7.2.3 / 7.2.53211 |
37
- | Triton distribution / imported version | 3.7.1+gitf0b55c07 / 3.7.1 |
38
- | Transformers | 5.17.0 |
39
- | Tokenizers / Safetensors | 0.23.2 / 0.8.0 |
40
- | NumPy / Einops | 2.3.5 / 0.8.2 |
41
- | Hugging Face Hub | 1.31.0 |
42
- | FLA core / Flash Linear Attention | 0.5.2 / 0.5.2 |
43
-
44
- `runtime-provenance.json` records the live registry manifest response, 36 shared base layers, selected actual binary SHA256 values, observed versions, wheel URLs and wheel hashes. The FLA wheels were downloaded and their SHA256 values matched the qualified overlay. `runtime-fla-requirements.lock` intentionally covers only that overlay; it is not a standalone dependency lock for an arbitrary system.
45
-
46
- The official [FLA installation guide](https://github.com/fla-org/flash-linear-attention/blob/main/INSTALL.md) separates backend PyTorch installation from FLA and documents `--no-deps` for pre-release/custom Torch builds. This recipe uses that boundary and never asks pip to replace Torch or Triton. Transformers is supplied by the pinned base; see its [official installation documentation](https://huggingface.co/docs/transformers/installation) for the general package installation model.
47
-
48
- ## Portability boundary
49
-
50
- The public [PyTorch ROCm 7.2 wheel index](https://download.pytorch.org/whl/rocm7.2/torch/) contains ordinary release wheels such as `2.12.0+rocm7.2`. They are a different artifact from the qualified `2.12.0+git6bbd260` build and are not interchangeable evidence. The latter's [source commit is public](https://github.com/pytorch/pytorch/commit/6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5), but a source commit alone does not reproduce compiler flags, linked libraries and binary behavior.
51
-
52
- An alternative runtime must recheck the installed versions, actual FLA Gated DeltaNet dispatch, BF16 backbone plus FP32 head, fixed prompt rendering, batch size eight, and model output agreement. Runtime changes can shift probabilities near a decision boundary even with identical weights. The supplied engine uses FLA Gated DeltaNet, reference PyTorch causal convolution and SDPA; selecting a different kernel is a new runtime configuration.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
RUNTIME_BINDING.json DELETED
@@ -1,76 +0,0 @@
1
- {
2
- "source_bundle_manifest_sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4",
3
- "profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
4
- "profile_validation_receipt_sha256": "7a66474325b5575fdd15a33e5e0340c19e809bdcc22adf4cbb45baa0e1b32693",
5
- "unchanged_files": [
6
- {
7
- "file": "backbone/config.json",
8
- "bytes": 1978,
9
- "sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
10
- },
11
- {
12
- "file": "backbone/model-00001-of-00003.safetensors",
13
- "bytes": 3991295368,
14
- "sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
15
- },
16
- {
17
- "file": "backbone/model-00002-of-00003.safetensors",
18
- "bytes": 3979828128,
19
- "sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
20
- },
21
- {
22
- "file": "backbone/model-00003-of-00003.safetensors",
23
- "bytes": 440425856,
24
- "sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
25
- },
26
- {
27
- "file": "backbone/model.safetensors.index.json",
28
- "bytes": 33047,
29
- "sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
30
- },
31
- {
32
- "file": "chat_template.jinja",
33
- "bytes": 7756,
34
- "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
35
- },
36
- {
37
- "file": "code/decision_model.py",
38
- "bytes": 10114,
39
- "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
40
- },
41
- {
42
- "file": "code/predict.py",
43
- "bytes": 3164,
44
- "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
45
- },
46
- {
47
- "file": "decision_config.json",
48
- "bytes": 753,
49
- "sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
50
- },
51
- {
52
- "file": "decision_head.safetensors",
53
- "bytes": 10529624,
54
- "sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
55
- },
56
- {
57
- "file": "temperature.json",
58
- "bytes": 6083,
59
- "sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
60
- },
61
- {
62
- "file": "tokenizer.json",
63
- "bytes": 19989325,
64
- "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
65
- },
66
- {
67
- "file": "tokenizer_config.json",
68
- "bytes": 1123,
69
- "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
70
- }
71
- ],
72
- "training_selection_calibration_unchanged": true,
73
- "recomputed_or_recalibrated": false,
74
- "private_compiled_cache_required": false,
75
- "new_bundle_offline_proof_required": true
76
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
SERVING_OPTIMIZATION.json DELETED
@@ -1,34 +0,0 @@
1
- {
2
- "format": "joint-serving-runtime-candidate-v1",
3
- "family": "Nox",
4
- "source_publication_revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654",
5
- "source_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
6
- "source_release_manifest_sha256": "3e11cc6011860ea245c53eb44b9da61a85d5feb5c6caa5da95428d63ede6d712",
7
- "source_api_sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40",
8
- "candidate_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
9
- "changed_inference_files": [
10
- "code/decision_api.py"
11
- ],
12
- "held_fixed": [
13
- "weights",
14
- "tokenizer",
15
- "prompt",
16
- "temperature",
17
- "normalization_profile",
18
- "BF16_backbone_FP32_head",
19
- "batch8",
20
- "input_limit16384",
21
- "public_wrapper"
22
- ],
23
- "shared_state_neural_cache": false,
24
- "cross_request_cache": false,
25
- "timing_evidence": {
26
- "analysis/decoder-joint-serving-v1/COMPLETED-PARITY.json": "4187fb76432eb69b263cb6d5ad55f10a09aef384fe405f3ad204acf1edc761b5",
27
- "analysis/decoder-joint-timing-v1/INDEPENDENTLY-REVIEWED.json": "1b416c8556c2305ffd18bd1e3cd523964820a5e2a3029cf447f20b2829ba82f7",
28
- "results/joint-serving-timing-v1/summary/SUMMARY.json": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
29
- "analysis/decoder4b/joint-timing-actual-review-v1/REVIEW.json": "e94fec9d9c6d487200ef7ea1cf82319b33c4836c08d83f0b856bf8590f2fbee6"
30
- },
31
- "candidate_default_entrypoint_proof_required": true,
32
- "published": false,
33
- "adoption_authorized": false
34
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
SOURCE_BUNDLE_MANIFEST.json DELETED
@@ -1,173 +0,0 @@
1
- {
2
- "format": "research-pointer-bundle-v1",
3
- "status": "candidate-export-awaiting-independent-reload-and-quality-gates",
4
- "files": [
5
- {
6
- "file": "backbone/config.json",
7
- "bytes": 1978,
8
- "sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
9
- },
10
- {
11
- "file": "backbone/model-00001-of-00003.safetensors",
12
- "bytes": 3991295368,
13
- "sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
14
- },
15
- {
16
- "file": "backbone/model-00002-of-00003.safetensors",
17
- "bytes": 3979828128,
18
- "sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
19
- },
20
- {
21
- "file": "backbone/model-00003-of-00003.safetensors",
22
- "bytes": 440425856,
23
- "sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
24
- },
25
- {
26
- "file": "backbone/model.safetensors.index.json",
27
- "bytes": 33047,
28
- "sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
29
- },
30
- {
31
- "file": "chat_template.jinja",
32
- "bytes": 7756,
33
- "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
34
- },
35
- {
36
- "file": "code/decision_api.py",
37
- "bytes": 6724,
38
- "sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21"
39
- },
40
- {
41
- "file": "code/decision_model.py",
42
- "bytes": 10114,
43
- "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
44
- },
45
- {
46
- "file": "code/predict.py",
47
- "bytes": 3164,
48
- "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
49
- },
50
- {
51
- "file": "decision_config.json",
52
- "bytes": 753,
53
- "sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
54
- },
55
- {
56
- "file": "decision_head.safetensors",
57
- "bytes": 10529624,
58
- "sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
59
- },
60
- {
61
- "file": "runtime.json",
62
- "bytes": 378,
63
- "sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9"
64
- },
65
- {
66
- "file": "temperature.json",
67
- "bytes": 6083,
68
- "sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
69
- },
70
- {
71
- "file": "tokenizer.json",
72
- "bytes": 19989325,
73
- "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
74
- },
75
- {
76
- "file": "tokenizer_config.json",
77
- "bytes": 1123,
78
- "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
79
- }
80
- ],
81
- "tensors": [
82
- {
83
- "file": "backbone/model-00001-of-00003.safetensors",
84
- "elements": 1995638528,
85
- "elements_by_dtype": {
86
- "BF16": 1995638528
87
- }
88
- },
89
- {
90
- "file": "backbone/model-00002-of-00003.safetensors",
91
- "elements": 1989900928,
92
- "elements_by_dtype": {
93
- "BF16": 1989900928
94
- }
95
- },
96
- {
97
- "file": "backbone/model-00003-of-00003.safetensors",
98
- "elements": 220211840,
99
- "elements_by_dtype": {
100
- "BF16": 220211840
101
- }
102
- },
103
- {
104
- "file": "decision_head.safetensors",
105
- "elements": 2632192,
106
- "elements_by_dtype": {
107
- "F32": 2632192
108
- }
109
- }
110
- ],
111
- "source_checkpoint_files": [
112
- {
113
- "file": "backbone/config.json",
114
- "bytes": 1977,
115
- "sha256": "a5ed4156fda05f0f9149c66964d6165916754e7355488a8a07d9b0398acdbdb9"
116
- },
117
- {
118
- "file": "backbone/model-00001-of-00005.safetensors",
119
- "bytes": 3992269864,
120
- "sha256": "d65565b8988a20b07988db09ed2321111e57b4fe656112fe4636715f8418c40b"
121
- },
122
- {
123
- "file": "backbone/model-00002-of-00005.safetensors",
124
- "bytes": 3990302320,
125
- "sha256": "99197c2df234da167190a18b34efe6275622687500346ad3d42d9fad6b25d52d"
126
- },
127
- {
128
- "file": "backbone/model-00003-of-00005.safetensors",
129
- "bytes": 3937871784,
130
- "sha256": "0d0af48a06e40b0afdcb7aa02cf6d8d6df885ccc9e2593e3db72cd3a92e798a7"
131
- },
132
- {
133
- "file": "backbone/model-00004-of-00005.safetensors",
134
- "bytes": 3926578128,
135
- "sha256": "b960486d5d833e016c9e2a32d42214db14a8e48afef49d61a7f514f8c636a264"
136
- },
137
- {
138
- "file": "backbone/model-00005-of-00005.safetensors",
139
- "bytes": 976029360,
140
- "sha256": "994dac35c408a89d89768febad1ed52a255da62f7924f86bd76371381e9cc751"
141
- },
142
- {
143
- "file": "decision_config.json",
144
- "bytes": 1070,
145
- "sha256": "3324b9318a873fc27a8610e19d6d7ba82b0890a49011abbc8b097bdf429e6747"
146
- },
147
- {
148
- "file": "decision_head.safetensors",
149
- "bytes": 10529624,
150
- "sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
151
- },
152
- {
153
- "file": "tokenizer.json",
154
- "bytes": 19989325,
155
- "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
156
- },
157
- {
158
- "file": "tokenizer_config.json",
159
- "bytes": 1123,
160
- "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
161
- }
162
- ],
163
- "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
164
- "source_api_code_sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21",
165
- "dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
166
- "production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
167
- "temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
168
- "runtime_sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9",
169
- "production_batch_size": 8,
170
- "input_length_limit": 16384,
171
- "original_checkpoint_name": "winner",
172
- "no_publication_performed": true
173
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
USAGE.md DELETED
@@ -1,59 +0,0 @@
1
- # Use Nox-4B
2
-
3
- The examples below use the [official TypeSafe SDK](https://docs.typesafe.ai/sdk/python/usage) and the standard [SystemOne HTTP request](https://docs.typesafe.ai/api). Configure your endpoint to serve `Decision-1.0-Nox-4B`, then replace the example URL and API key. A Hugging Face model repository is a weights download, not an inference endpoint.
4
-
5
- ```bash
6
- pip install typesafe-sdk
7
- ```
8
-
9
- ```python
10
- from typesafe_sdk import Choice, Noul, TypeSafeClient
11
-
12
- with TypeSafeClient(
13
- api_key="YOUR_ENDPOINT_API_KEY",
14
- base_url="https://your-decision-endpoint.example",
15
- model="Decision-1.0-Nox-4B",
16
- ) as client:
17
- result = client.system_one(
18
- state="Customer reports a duplicate charge and asks for a refund.",
19
- questions={
20
- "route": Choice(
21
- instructions="Which team should handle this request?",
22
- criteria={"billing": "Payments and refunds", "technical": "Product faults"},
23
- ),
24
- "refund_requested": Noul(instructions="Did the customer request a refund?"),
25
- },
26
- )
27
- print(result.choices["route"].choice)
28
- print(result.nouls["refund_requested"].noul)
29
- ```
30
-
31
- ```bash
32
- curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
33
- -H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
34
- -H 'Content-Type: application/json' \
35
- --data-raw '{
36
- "model": "Decision-1.0-Nox-4B",
37
- "state": "Customer reports a duplicate charge and asks for a refund.",
38
- "questions": {
39
- "route": {
40
- "type": "choice",
41
- "instructions": "Which team should handle this request?",
42
- "criteria": {
43
- "billing": "Payments and refunds",
44
- "technical": "Product faults"
45
- }
46
- },
47
- "refund_requested": {
48
- "type": "noul",
49
- "instructions": "Did the customer request a refund?"
50
- }
51
- }
52
- }'
53
- ```
54
-
55
- The state can be text or JSON-compatible structured data. Question IDs and Choice IDs are preserved in the response. Choice uses 2–255 options; Noul returns `noul`, the probability of a condition being true; Score uses 2–10 rubric descriptions ordered from index zero. A request can contain many questions.
56
-
57
- Choice returns `choice`, `probabilities` and `confidence`. Score returns an expected zero-based `score`, a probability distribution and `legend`. Local Decision confidence is normalized maximum probability, `(K × max(p) − 1)/(K − 1)`; it does not reproduce an unpublished provider confidence statistic. An HTTP integration must supply `usage.input_tokens` and `usage.output_tokens`; counting generated tokens as zero is appropriate for this non-generative model, not a claim about provider billing.
58
-
59
- The native runtime processes independent complete questions in batches of eight. Each complete rendered question, including state, instructions and criteria, must fit 16,384 tokens; overflow is rejected. [Runtime and hardware requirements](RUNTIME.md).
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
WEIGHTING.md DELETED
@@ -1,21 +0,0 @@
1
- # Weight sensitivity
2
-
3
- The current product-priority weights were chosen after observing results. This comparison holds every model and prediction fixed; reweighting is not a training improvement.
4
-
5
- | Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
6
- |---|---:|---:|---:|
7
- | Lux-9B | 77.40 | 77.07 | 79.69 |
8
- | Nox-4B | 73.09 | 72.42 | 75.03 |
9
- | Kev-9B | 71.89 | 72.01 | 73.19 |
10
- | Kev-4B | 70.09 | 70.30 | 71.73 |
11
- | Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
12
- | Decider | 67.71 | 67.97 | 71.75 |
13
- | Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
14
- | Sol-2B | 66.32 | 65.48 | 70.14 |
15
- | Eos-0.8B | 61.89 | 61.19 | 65.99 |
16
- | Kev-0.8B | 58.28 | 58.33 | 59.75 |
17
- | Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
18
- | Kai-0.6B | 53.52 | 53.05 | 55.82 |
19
- | Laya · English | 51.03 | 50.85 | 51.76 |
20
- | Laya · Multilingual | 47.19 | 47.18 | 48.56 |
21
- | Jev | 81.05 | 81.45 | 82.45 |
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
assets/decision-expanded-old_core-600px.png DELETED

Git LFS Details

  • SHA256: f1586b4bed9423dba1ce138a343ba35de7f95a477aa0946031e99f8c8950279a
  • Pointer size: 131 Bytes
  • Size of remote file: 168 kB
assets/decision-expanded-old_core.pdf DELETED
Binary file (44.4 kB)
 
assets/decision-expanded-old_core.png DELETED

Git LFS Details

  • SHA256: e662de90db73e029550f5badd9a32cc2adb2e3186eb96b12940994bb3ed9eac1
  • Pointer size: 131 Bytes
  • Size of remote file: 312 kB
assets/decision-expanded-old_core.svg DELETED
assets/decision-expanded-overview-600px.png DELETED
Binary file (94.9 kB)
 
assets/decision-expanded-overview.pdf DELETED
Binary file (41.6 kB)
 
assets/decision-expanded-overview.png DELETED

Git LFS Details

  • SHA256: c671be3a92da5bb920864d7b0cdbd6ec22e24213d66052e165c20363b319e7e5
  • Pointer size: 131 Bytes
  • Size of remote file: 187 kB
assets/decision-expanded-overview.svg DELETED
assets/decision-expanded-ranking-600px.png DELETED
Binary file (83.3 kB)
 
assets/decision-expanded-ranking.pdf DELETED
Binary file (37.1 kB)
 
assets/decision-expanded-ranking.png DELETED

Git LFS Details

  • SHA256: e098abf598e3aa51dbef517ed7aeb6c2f7d9312d13bef37d5dbd40ce1dd76aa4
  • Pointer size: 131 Bytes
  • Size of remote file: 186 kB
assets/decision-expanded-ranking.svg DELETED
assets/decision-expanded-v3_core-600px.png DELETED

Git LFS Details

  • SHA256: 0630ac755fbfb100ebb3ecb965a5ebe34f511384b1a4c98468d6f0fdbc239d13
  • Pointer size: 131 Bytes
  • Size of remote file: 168 kB
assets/decision-expanded-v3_core.pdf DELETED
Binary file (44.5 kB)
 
assets/decision-expanded-v3_core.png DELETED

Git LFS Details

  • SHA256: d5eaa30f40e03e06414650c023c4994c13182a6847ec423155a9e126df1d16a0
  • Pointer size: 131 Bytes
  • Size of remote file: 300 kB
assets/decision-expanded-v3_core.svg DELETED
assets/decision-expanded-v4-600px.png DELETED
Binary file (76.7 kB)
 
assets/decision-expanded-v4.pdf DELETED
Binary file (37.7 kB)
 
assets/decision-expanded-v4.png DELETED

Git LFS Details

  • SHA256: 47f5d01cea0c8d620f9076f8972bf3457c7e234c99ddcb734250ea3b118e78a9
  • Pointer size: 131 Bytes
  • Size of remote file: 153 kB
assets/decision-expanded-v4.svg DELETED
assets/decision-expanded-v5-600px.png DELETED
Binary file (95.1 kB)
 
assets/decision-expanded-v5.pdf DELETED
Binary file (40.2 kB)
 
assets/decision-expanded-v5.png DELETED

Git LFS Details

  • SHA256: 057fe35e93ebe41f3bb8653392c02de27616aa248fb6e2a25026f3d2b1ba9f5a
  • Pointer size: 131 Bytes
  • Size of remote file: 188 kB
assets/decision-expanded-v5.svg DELETED
assets/decision-family-header.png DELETED

Git LFS Details

  • SHA256: 213511289ce8df038d938ac470e803c427ed57f0f85cc397dd4d79964b866541
  • Pointer size: 132 Bytes
  • Size of remote file: 2.96 MB
assets/decision-matrix.pdf DELETED
Binary file (29.4 kB)
 
assets/decision-matrix.svg DELETED
assets/decision-question-scaling-600px.png DELETED
Binary file (40.8 kB)
 
assets/decision-question-scaling.pdf DELETED
Binary file (19 kB)
 
assets/decision-question-scaling.svg DELETED
assets/decision-ranking.pdf DELETED
Binary file (25.4 kB)
 
assets/decision-ranking.svg DELETED
bundle-manifest.json DELETED
@@ -1,243 +0,0 @@
1
- {
2
- "format": "research-pointer-bundle-v1",
3
- "status": "null-description-candidate-awaiting-public-proof-and-full-regression",
4
- "files": [
5
- {
6
- "file": "NORMALIZATION_RUNTIME.md",
7
- "bytes": 756,
8
- "sha256": "cf8e6ce1f07687a68b6adeb98e6704b8f292cb6adc7b10e4b84e8e70aa5472e5"
9
- },
10
- {
11
- "file": "NULL_DESCRIPTION_RENDERING.json",
12
- "bytes": 514,
13
- "sha256": "3b531cab60cba35648ab71fe4be7fd1af619f02ee337e3beed6296fdfc8b9ec8"
14
- },
15
- {
16
- "file": "RUNTIME_BINDING.json",
17
- "bytes": 2621,
18
- "sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
19
- },
20
- {
21
- "file": "SERVING_OPTIMIZATION.json",
22
- "bytes": 1548,
23
- "sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
24
- },
25
- {
26
- "file": "SOURCE_BUNDLE_MANIFEST.json",
27
- "bytes": 5637,
28
- "sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4"
29
- },
30
- {
31
- "file": "backbone/config.json",
32
- "bytes": 1978,
33
- "sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
34
- },
35
- {
36
- "file": "backbone/model-00001-of-00003.safetensors",
37
- "bytes": 3991295368,
38
- "sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
39
- },
40
- {
41
- "file": "backbone/model-00002-of-00003.safetensors",
42
- "bytes": 3979828128,
43
- "sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
44
- },
45
- {
46
- "file": "backbone/model-00003-of-00003.safetensors",
47
- "bytes": 440425856,
48
- "sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
49
- },
50
- {
51
- "file": "backbone/model.safetensors.index.json",
52
- "bytes": 33047,
53
- "sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
54
- },
55
- {
56
- "file": "chat_template.jinja",
57
- "bytes": 7756,
58
- "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
59
- },
60
- {
61
- "file": "code/decision_api.py",
62
- "bytes": 10978,
63
- "sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e"
64
- },
65
- {
66
- "file": "code/decision_model.py",
67
- "bytes": 10114,
68
- "sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
69
- },
70
- {
71
- "file": "code/predict.py",
72
- "bytes": 3164,
73
- "sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
74
- },
75
- {
76
- "file": "code/profile_guard.py",
77
- "bytes": 6184,
78
- "sha256": "061d6ba3edf032038b074b061879a1ff81cfd2928fc0131c66b15563f7822509"
79
- },
80
- {
81
- "file": "code/runtime_profile.py",
82
- "bytes": 2857,
83
- "sha256": "3e26ed65f2cd706def209761421cb8fc825281e5c1ed8f168350365ad803dc96"
84
- },
85
- {
86
- "file": "decision_config.json",
87
- "bytes": 753,
88
- "sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
89
- },
90
- {
91
- "file": "decision_head.safetensors",
92
- "bytes": 10529624,
93
- "sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
94
- },
95
- {
96
- "file": "pyproject.toml",
97
- "bytes": 436,
98
- "sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946"
99
- },
100
- {
101
- "file": "runtime-profile/l2norm_fwd_kernel.json",
102
- "bytes": 26330,
103
- "sha256": "d7ed7c9962a48efdfa76ed695c8bdc0f56afe1397c202c936eb78315e87a26df"
104
- },
105
- {
106
- "file": "runtime-profile/profile.json",
107
- "bytes": 35040,
108
- "sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850"
109
- },
110
- {
111
- "file": "runtime.json",
112
- "bytes": 1112,
113
- "sha256": "78a2c21f8138beeffb3e6b1c11e03cb48972a78ca5ef3d1f58c50651031c34b6"
114
- },
115
- {
116
- "file": "src/decision/__init__.py",
117
- "bytes": 167,
118
- "sha256": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313"
119
- },
120
- {
121
- "file": "src/decision/example.py",
122
- "bytes": 3581,
123
- "sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6"
124
- },
125
- {
126
- "file": "src/decision/model.py",
127
- "bytes": 9415,
128
- "sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0"
129
- },
130
- {
131
- "file": "temperature.json",
132
- "bytes": 6083,
133
- "sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
134
- },
135
- {
136
- "file": "tokenizer.json",
137
- "bytes": 19989325,
138
- "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
139
- },
140
- {
141
- "file": "tokenizer_config.json",
142
- "bytes": 1123,
143
- "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
144
- }
145
- ],
146
- "tensors": [
147
- {
148
- "file": "backbone/model-00001-of-00003.safetensors",
149
- "elements": 1995638528,
150
- "elements_by_dtype": {
151
- "BF16": 1995638528
152
- }
153
- },
154
- {
155
- "file": "backbone/model-00002-of-00003.safetensors",
156
- "elements": 1989900928,
157
- "elements_by_dtype": {
158
- "BF16": 1989900928
159
- }
160
- },
161
- {
162
- "file": "backbone/model-00003-of-00003.safetensors",
163
- "elements": 220211840,
164
- "elements_by_dtype": {
165
- "BF16": 220211840
166
- }
167
- },
168
- {
169
- "file": "decision_head.safetensors",
170
- "elements": 2632192,
171
- "elements_by_dtype": {
172
- "F32": 2632192
173
- }
174
- }
175
- ],
176
- "source_checkpoint_files": [
177
- {
178
- "file": "backbone/config.json",
179
- "bytes": 1977,
180
- "sha256": "a5ed4156fda05f0f9149c66964d6165916754e7355488a8a07d9b0398acdbdb9"
181
- },
182
- {
183
- "file": "backbone/model-00001-of-00005.safetensors",
184
- "bytes": 3992269864,
185
- "sha256": "d65565b8988a20b07988db09ed2321111e57b4fe656112fe4636715f8418c40b"
186
- },
187
- {
188
- "file": "backbone/model-00002-of-00005.safetensors",
189
- "bytes": 3990302320,
190
- "sha256": "99197c2df234da167190a18b34efe6275622687500346ad3d42d9fad6b25d52d"
191
- },
192
- {
193
- "file": "backbone/model-00003-of-00005.safetensors",
194
- "bytes": 3937871784,
195
- "sha256": "0d0af48a06e40b0afdcb7aa02cf6d8d6df885ccc9e2593e3db72cd3a92e798a7"
196
- },
197
- {
198
- "file": "backbone/model-00004-of-00005.safetensors",
199
- "bytes": 3926578128,
200
- "sha256": "b960486d5d833e016c9e2a32d42214db14a8e48afef49d61a7f514f8c636a264"
201
- },
202
- {
203
- "file": "backbone/model-00005-of-00005.safetensors",
204
- "bytes": 976029360,
205
- "sha256": "994dac35c408a89d89768febad1ed52a255da62f7924f86bd76371381e9cc751"
206
- },
207
- {
208
- "file": "decision_config.json",
209
- "bytes": 1070,
210
- "sha256": "3324b9318a873fc27a8610e19d6d7ba82b0890a49011abbc8b097bdf429e6747"
211
- },
212
- {
213
- "file": "decision_head.safetensors",
214
- "bytes": 10529624,
215
- "sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
216
- },
217
- {
218
- "file": "tokenizer.json",
219
- "bytes": 19989325,
220
- "sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
221
- },
222
- {
223
- "file": "tokenizer_config.json",
224
- "bytes": 1123,
225
- "sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
226
- }
227
- ],
228
- "source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
229
- "source_api_code_sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e",
230
- "dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
231
- "production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
232
- "temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
233
- "runtime_sha256": "78a2c21f8138beeffb3e6b1c11e03cb48972a78ca5ef3d1f58c50651031c34b6",
234
- "production_batch_size": 8,
235
- "input_length_limit": 16384,
236
- "original_checkpoint_name": "winner",
237
- "no_publication_performed": true,
238
- "source_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
239
- "normalization_profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
240
- "public_wrapper_included": true,
241
- "source_published_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
242
- "serving_optimization_sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
243
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
code/decision_api.py DELETED
@@ -1,198 +0,0 @@
1
- """Typed local inference adapter for the research decision checkpoints.
2
-
3
- The response schema resembles TypeSafe's primitives. Confidence uses this
4
- implementation's documented normalized maximum probability, not a claimed
5
- reimplementation of TypeSafe's unpublished statistic. No text generation.
6
- """
7
- from __future__ import annotations
8
- import importlib.util
9
- import math
10
- from pathlib import Path
11
-
12
-
13
- def prepare_runtime_profile(checkpoint, device='cuda:0'):
14
- # This profile is verified before any dependency import can choose kernels.
15
- import hashlib, json, sys
16
- root=Path(checkpoint);runtime=json.loads((root/'runtime.json').read_text())
17
- spec=runtime.get('normalization_profile')
18
- if spec is None:
19
- if '_decision_process_normalization_profile_v1' in sys.modules:
20
- raise RuntimeError('Use separate processes for profiled and unprofiled models')
21
- return None
22
- import torch
23
- target=torch.device(device)
24
- if target.type!='cuda' or not torch.cuda.is_available():
25
- raise RuntimeError('The bound profile requires a ROCm CUDA device')
26
- arch=getattr(torch.cuda.get_device_properties(target),'gcnArchName','').split(':')[0]
27
- if arch!=spec['validated_arch']:
28
- raise RuntimeError('Target GPU architecture does not match the bound profile: '+arch)
29
- relative=Path(spec['loader_file'])
30
- if relative.is_absolute() or '..' in relative.parts:raise ValueError('Unsafe profile loader path')
31
- path=root/relative
32
- if hashlib.sha256(path.read_bytes()).hexdigest()!=spec['loader_sha256']:
33
- raise ValueError('Bound runtime profile loader changed')
34
- definition=importlib.util.spec_from_file_location('decision_bundle_runtime_profile',path)
35
- module=importlib.util.module_from_spec(definition);definition.loader.exec_module(module)
36
- return module.ensure_profile(root)
37
-
38
-
39
- def question_row(state, name, question):
40
- kind=question.get('type')
41
- if kind not in {'choice','noul','score'}:raise ValueError('Unknown question type')
42
- if 'instructions' not in question:raise ValueError('instructions is required')
43
- criteria=question.get('criteria')
44
- if kind=='noul':
45
- criteria={} if criteria is None else criteria
46
- if not isinstance(criteria,dict) or set(criteria)-{'true','false'}:
47
- raise ValueError('noul criteria may contain only true and false')
48
- options=[{'key':'false','description':criteria.get('false','The answer to the question is no.')},
49
- {'key':'true','description':criteria.get('true','The answer to the question is yes.')}]
50
- elif kind=='score':
51
- if not isinstance(criteria,list) or not 2<=len(criteria)<=10:
52
- raise ValueError('score requires an ordered list of 2..10 criteria')
53
- options=[{'key':str(i),'description':value} for i,value in enumerate(criteria)]
54
- else:
55
- if not isinstance(criteria,dict) or not 2<=len(criteria)<=255:
56
- raise ValueError('choice requires a mapping of 2..255 criteria')
57
- if not all(isinstance(k,str) for k in criteria):raise ValueError('Choice keys must be strings')
58
- options=[{'key':key,'description':key if value is None else value} for key,value in criteria.items()]
59
- # The question name is used for bookkeeping only; encoders never render id.
60
- return {'id':name,'state':state,'instructions':question['instructions'],
61
- 'options':options,'task_type':kind,'family':'inference'}
62
-
63
-
64
- def typed_answer(row, probabilities):
65
- p=[float(v) for v in probabilities];k=len(row['options'])
66
- if len(p)!=k or any(not math.isfinite(v) or v<0 for v in p):
67
- raise ValueError('Invalid probability vector')
68
- total=sum(p)
69
- if total<=0 or abs(total-1)>1e-4:raise ValueError('Probabilities must sum to one')
70
- p=[v/total for v in p];selected=max(range(k),key=p.__getitem__)
71
- kind=row['task_type']
72
- if kind=='noul':
73
- keys=[o['key'] for o in row['options']]
74
- if set(keys)!={'false','true'}:raise ValueError('Native noul rows require false/true keys')
75
- return {'type':'noul','noul':p[keys.index('true')]}
76
- answer={'type':kind,'probabilities':{o['key']:v for o,v in zip(row['options'],p)},
77
- 'confidence':max(0.,min(1.,(k*max(p)-1)/(k-1)))}
78
- if kind=='choice':answer['choice']=row['options'][selected]['key']
79
- else:
80
- if [o['key'] for o in row['options']] != [str(i) for i in range(k)]:
81
- raise ValueError('Native score rows require ordered numeric level keys')
82
- answer['score']=sum(i*v for i,v in enumerate(p))
83
- answer['legend']={str(i):o['description'] for i,o in enumerate(row['options'])}
84
- return answer
85
-
86
-
87
- class DecisionEngine:
88
- def __init__(self, checkpoint, model_code, *, device='cuda:0', max_length=16384,
89
- batch_size=8, temperatures=None, model_name='local-decision-research'):
90
- self.normalization_profile=prepare_runtime_profile(checkpoint, device=device)
91
- import torch
92
- path=Path(model_code)/'decision_model.py'
93
- spec=importlib.util.spec_from_file_location('research_decision_runtime',path)
94
- module=importlib.util.module_from_spec(spec);spec.loader.exec_module(module)
95
- model,tokenizer=module.DecisionModel.from_checkpoint(checkpoint,dtype=torch.bfloat16)
96
- self.model=model.to(device).eval();self.tokenizer=tokenizer;self.module=module
97
- self.device=device;self.max_length=max_length;self.batch_size=batch_size
98
- self.temperatures=temperatures or {};self.model_name=model_name
99
- if batch_size<1 or max_length<1:raise ValueError('Positive batch_size/max_length required')
100
- if any(not math.isfinite(v) or v<=0 for v in self.temperatures.values()):
101
- raise ValueError('Temperatures must be finite positive numbers')
102
-
103
- def predict_rows(self, rows):
104
- import torch
105
- encoded=encode_request(rows,self.tokenizer,self.module,self.max_length)
106
- pad=self.tokenizer.pad_token_id if self.tokenizer.pad_token_id is not None else self.tokenizer.eos_token_id
107
- records=[]
108
- with torch.inference_mode():
109
- for start in range(0,len(rows),self.batch_size):
110
- items=encoded[start:start+self.batch_size]
111
- batch={key:value.to(self.device) if torch.is_tensor(value) else value
112
- for key,value in self.module.collate(items,pad).items()}
113
- with torch.autocast('cuda',dtype=torch.bfloat16):logits=self.model(**batch)
114
- # Preserve each row's original float/temperature/softmax math,
115
- # but defer host synchronization until the complete batch.
116
- staged=[];transfers=[]
117
- for row,item,values in zip(rows[start:start+self.batch_size],items,logits):
118
- k=len(row['options']);values=values[:k].float()
119
- temperature=self.temperatures.get(row['task_type'],1.)
120
- probabilities=(values/temperature).softmax(-1)
121
- staged.append((row,item,k,temperature))
122
- transfers.extend((values,probabilities))
123
- host_values=torch.cat(transfers).tolist()
124
- offset=0
125
- for row,item,k,temperature in staged:
126
- values=host_values[offset:offset+k]
127
- probabilities=host_values[offset+k:offset+2*k]
128
- offset+=2*k
129
- answer=typed_answer(row,probabilities)
130
- prediction=max(range(k),key=probabilities.__getitem__)
131
- if row['task_type']=='noul':
132
- chosen='true' if answer['noul']>=.5 else 'false'
133
- prediction=[o['key'] for o in row['options']].index(chosen)
134
- rec={'id':row['id'],'status':'ok','prediction':prediction,
135
- 'probabilities':probabilities,'logits':values,'temperature':temperature,
136
- 'native_contract':True,'truncated':False,'input_tokens':len(item['ids']),
137
- 'prompt_sha256':item['prompt_sha256'],'answer':answer}
138
- if row['task_type']=='noul':rec['native_noul']=answer['noul']
139
- if row['task_type']=='score':rec['native_score']=answer['score']
140
- records.append(rec)
141
- return records
142
-
143
- def decide(self, state, questions):
144
- if not isinstance(questions,dict) or not questions:
145
- raise ValueError('questions must be a nonempty mapping')
146
- if not all(isinstance(name,str) for name in questions):raise ValueError('Question names must be strings')
147
- rows=[question_row(state,name,q) for name,q in questions.items()]
148
- result=self.predict_rows(rows)
149
- return {'model':self.model_name,'answers':{r['id']:r['answer'] for r in result},
150
- 'usage':{'input_tokens':sum(r['input_tokens'] for r in result),'scored_questions':len(result)}}
151
-
152
-
153
- """Experimental request-local exact-segment tokenization.
154
-
155
- Original encode/segments functions remain authoritative. Batch tokenize exact
156
- whole segments, never split a BPE prefix at a new boundary. No cross-request
157
- cache, GPU change, prompt change or change to the eight-row inference groups.
158
- """
159
-
160
-
161
- class SegmentLookup:
162
- def __init__(self, tokenizer, cache):
163
- self.tokenizer, self.cache = tokenizer, cache
164
-
165
- def encode(self, text, **kwargs):
166
- if kwargs == {'add_special_tokens': False} and text in self.cache:
167
- # Original encode extends its prefix list in place.
168
- return list(self.cache[text])
169
- return self.tokenizer.encode(text, **kwargs)
170
-
171
-
172
- def encode_request(rows, tokenizer, module, max_length=16384,
173
- max_cached_characters=8_000_000, segment_batch_size=64):
174
- if max_cached_characters < 0 or segment_batch_size < 1:
175
- raise ValueError('Invalid tokenizer resource bound')
176
- unique = {}
177
- characters = 0
178
- for row in rows:
179
- prefix, options, suffix = module.segments(row)
180
- for segment in (prefix, *options, suffix):
181
- if segment not in unique:
182
- unique[segment] = None
183
- characters += len(segment)
184
- if characters > max_cached_characters:
185
- # Preserve the original behavior under the resource cap.
186
- return [module.encode(r, tokenizer, max_length) for r in rows]
187
- strings = list(unique)
188
- for start in range(0, len(strings), segment_batch_size):
189
- batch = strings[start:start + segment_batch_size]
190
- result = tokenizer(batch, add_special_tokens=False, padding=False,
191
- truncation=False, return_attention_mask=False,
192
- return_token_type_ids=False)['input_ids']
193
- if len(result) != len(batch):
194
- raise ValueError('Batch tokenizer output count differs')
195
- for segment, ids in zip(batch, result):
196
- unique[segment] = tuple(ids)
197
- lookup = SegmentLookup(tokenizer, unique)
198
- return [module.encode(row, lookup, max_length) for row in rows]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
code/decision_model.py DELETED
@@ -1,173 +0,0 @@
1
- """Dynamic candidate readout over a causal Qwen3.5 text backbone.
2
-
3
- Candidate endpoints retain their contextual vectors. A final global-query
4
- vector can incorporate all options before a shared bilinear + MLP scorer
5
- scores every candidate. This is a research architecture, not a Jev claim.
6
- """
7
- import hashlib
8
- import json
9
- import math
10
- from pathlib import Path
11
-
12
- import torch
13
- from torch import nn
14
- import torch.nn.functional as F
15
- from safetensors.torch import load_file, save_file
16
- from transformers import AutoTokenizer, Qwen3_5ForConditionalGeneration
17
- from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5TextModel
18
-
19
- PROMPT_VERSION = "structured-segmented-candidate-endpoints-global-query-v2"
20
- MAX_OPTIONS = 255
21
-
22
-
23
- def canonical(value):
24
- return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
25
-
26
-
27
- def payload(value):
28
- return value if isinstance(value, str) else canonical(value)
29
-
30
-
31
- def segments(row):
32
- opts = row["options"]
33
- if not 2 <= len(opts) <= MAX_OPTIONS:
34
- raise ValueError(f"{row['id']}: expected 2..255 options")
35
- if not all(isinstance(o["key"], str) for o in opts):
36
- raise ValueError("Option keys must be strings")
37
- if len({o["key"] for o in opts}) != len(opts):
38
- raise ValueError("Duplicate option keys")
39
- prefix = f"Context:\n{payload(row['state'])}\n\nTask type: {row.get('task_type', 'choice')}\nQuestion:\n{payload(row['instructions'])}\nOptions:"
40
- # Tokenize each part separately. This deliberately fixes boundaries and
41
- # avoids guessing endpoint indices from merged BPE character offsets.
42
- options = ["\n<option>\n" + canonical({"key": o["key"], "description": o.get("description")}) + "\n</option>" for o in opts]
43
- suffix = "\n\nSelect the single option best supported by the context and instructions.\nDecision:"
44
- return prefix, options, suffix
45
-
46
-
47
- def render(row):
48
- prefix, opts, suffix = segments(row)
49
- return prefix + "".join(opts) + suffix
50
-
51
-
52
- def encode(row, tokenizer, max_length=16384):
53
- prefix, opts, suffix = segments(row)
54
- ids = tokenizer.encode(prefix, add_special_tokens=False)
55
- candidate_positions = []
56
- for option in opts:
57
- part = tokenizer.encode(option, add_special_tokens=False)
58
- if not part:
59
- raise ValueError("Empty tokenized candidate")
60
- ids.extend(part)
61
- candidate_positions.append(len(ids) - 1)
62
- ids.extend(tokenizer.encode(suffix, add_special_tokens=False))
63
- if len(ids) > max_length:
64
- raise ValueError(f"{row['id']}: {len(ids)} tokens exceeds max_length={max_length}; no truncation allowed")
65
- label = row.get("label", -1)
66
- if label != -1 and not 0 <= label < len(opts):
67
- raise ValueError("Invalid label")
68
- prompt = prefix + "".join(opts) + suffix
69
- return {"id": row["id"], "ids": ids, "label": label, "nopts": len(opts), "family": row.get("family", "unspecified"), "candidate_positions": candidate_positions, "query_position": len(ids) - 1, "target_probs": row.get("target_probs"), "prompt_sha256": hashlib.sha256(prompt.encode()).hexdigest(), "token_ids_sha256": hashlib.sha256(canonical(ids).encode()).hexdigest(), "segmented_tokenization": True}
70
-
71
-
72
- def collate(items, pad_id):
73
- length = ((max(len(x["ids"]) for x in items) + 31) // 32) * 32
74
- nopts = max(x["nopts"] for x in items)
75
- ids = torch.full((len(items), length), pad_id, dtype=torch.long)
76
- mask = torch.zeros_like(ids)
77
- positions = torch.zeros((len(items), nopts), dtype=torch.long)
78
- candidate_mask = torch.zeros((len(items), nopts), dtype=torch.bool)
79
- for i, item in enumerate(items):
80
- if len(item["candidate_positions"]) != item["nopts"]:
81
- raise ValueError("Candidate count does not match endpoint count")
82
- if not all(0 <= p < item["query_position"] < len(item["ids"]) for p in item["candidate_positions"]):
83
- raise ValueError("Candidate endpoints must precede global query")
84
- if len(set(item["candidate_positions"])) != item["nopts"]:
85
- raise ValueError("Duplicate candidate endpoint")
86
- ids[i, :len(item["ids"])] = torch.tensor(item["ids"])
87
- mask[i, :len(item["ids"])] = 1
88
- positions[i, :item["nopts"]] = torch.tensor(item["candidate_positions"])
89
- candidate_mask[i, :item["nopts"]] = True
90
- return {"input_ids": ids, "attention_mask": mask, "candidate_positions": positions, "candidate_mask": candidate_mask, "query_positions": torch.tensor([x["query_position"] for x in items]), "labels": torch.tensor([x["label"] for x in items]), "nopts": torch.tensor([x["nopts"] for x in items]), "ids": [x["id"] for x in items], "families": [x["family"] for x in items]}
91
-
92
-
93
- class CandidateHead(nn.Module):
94
- def __init__(self, hidden_size, head_dim=256):
95
- super().__init__()
96
- self.head_dim = head_dim
97
- self.candidate_norm = nn.LayerNorm(hidden_size)
98
- self.query_norm = nn.LayerNorm(hidden_size)
99
- self.key = nn.Linear(hidden_size, head_dim, bias=False)
100
- self.query = nn.Linear(hidden_size, head_dim, bias=False)
101
- self.candidate_mlp = nn.Linear(hidden_size, head_dim, bias=True)
102
- self.query_mlp = nn.Linear(hidden_size, head_dim, bias=False)
103
- self.scalar = nn.Linear(head_dim, 1, bias=False)
104
- nn.init.normal_(self.scalar.weight, mean=0., std=0.01)
105
-
106
- def forward(self, candidates, query):
107
- # Keep the small shared head in FP32 even when the backbone uses BF16.
108
- # The v1 letter head exhibited BF16 ties sensitive to batch padding.
109
- with torch.autocast(device_type=candidates.device.type, enabled=False):
110
- c = self.candidate_norm(candidates.float())
111
- q = self.query_norm(query.float())
112
- bilinear = (self.key(c) * self.query(q)[:, None, :]).sum(-1) / math.sqrt(self.head_dim)
113
- interaction = self.scalar(F.gelu(self.candidate_mlp(c) + self.query_mlp(q)[:, None, :])).squeeze(-1)
114
- return bilinear + interaction
115
-
116
-
117
- class DecisionModel(nn.Module):
118
- def __init__(self, backbone, head, metadata):
119
- super().__init__()
120
- self.backbone, self.head, self.metadata = backbone, head, metadata
121
-
122
- @classmethod
123
- def from_base(cls, path, revision="local", dtype=torch.bfloat16, attention="sdpa", head_dim=256):
124
- tokenizer = AutoTokenizer.from_pretrained(path, local_files_only=True)
125
- full, info = Qwen3_5ForConditionalGeneration.from_pretrained(path, dtype=dtype, local_files_only=True, attn_implementation=attention, output_loading_info=True)
126
- if any(info.get(k) for k in ("missing_keys", "mismatched_keys", "error_msgs")):
127
- raise RuntimeError(f"Incomplete base loading: {info}")
128
- backbone = full.model.language_model
129
- backbone.config.use_cache = False
130
- head = CandidateHead(backbone.config.hidden_size, head_dim)
131
- metadata = {"base_revision": revision, "text_parameter_count": sum(p.numel() for p in backbone.parameters()), "prompt_version": PROMPT_VERSION, "attention": attention, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "head_precision": "float32-outside-autocast"}
132
- return cls(backbone, head, metadata), tokenizer
133
-
134
- @classmethod
135
- def from_decision_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa", head_dim=256):
136
- """Warm-start the backbone of a trained v1 model; initialize a new head."""
137
- path = Path(path)
138
- metadata = json.loads((path / "decision_config.json").read_text())
139
- backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention)
140
- backbone.config.use_cache = False
141
- head = CandidateHead(backbone.config.hidden_size, head_dim)
142
- metadata.update({"prompt_version": PROMPT_VERSION, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "warm_start": "trained-v1-text-backbone", "head_precision": "float32-outside-autocast"})
143
- return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True)
144
-
145
- @classmethod
146
- def from_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa"):
147
- path = Path(path)
148
- metadata = json.loads((path / "decision_config.json").read_text())
149
- if metadata["prompt_version"] != PROMPT_VERSION:
150
- raise ValueError("Not a pointer-v2 checkpoint; use from_decision_checkpoint for warm start")
151
- backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention)
152
- head = CandidateHead(backbone.config.hidden_size, metadata["head_dim"])
153
- head.load_state_dict(load_file(path / "decision_head.safetensors"))
154
- return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True)
155
-
156
- def forward(self, input_ids, attention_mask, candidate_positions, candidate_mask, query_positions, **unused):
157
- hidden = self.backbone(input_ids=input_ids, attention_mask=attention_mask, use_cache=False).last_hidden_state
158
- batches = torch.arange(hidden.shape[0], device=hidden.device)
159
- candidates = hidden[batches[:, None], candidate_positions]
160
- query = hidden[batches, query_positions]
161
- scores = self.head(candidates, query).float()
162
- return scores.masked_fill(~candidate_mask, -float("inf"))
163
-
164
- def save(self, path, tokenizer):
165
- path = Path(path); path.mkdir(parents=True, exist_ok=True)
166
- self.backbone.save_pretrained(path / "backbone", safe_serialization=True, max_shard_size="4GB")
167
- save_file({n: v.detach().cpu().contiguous() for n, v in self.head.state_dict().items()}, str(path / "decision_head.safetensors"))
168
- tokenizer.save_pretrained(path)
169
- (path / "decision_config.json").write_text(json.dumps(self.metadata, indent=2) + "\n")
170
-
171
-
172
- def classification_loss(logits, labels):
173
- return F.cross_entropy(logits, labels)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
code/predict.py DELETED
@@ -1,43 +0,0 @@
1
- """Portable one-GPU JSONL inference for frozen dynamic-option checkpoints."""
2
- import argparse
3
- import hashlib
4
- import json
5
- from pathlib import Path
6
- import time
7
- import torch
8
- from decision_model import DecisionModel, encode, collate
9
-
10
- p = argparse.ArgumentParser()
11
- p.add_argument("--model", required=True)
12
- p.add_argument("--base", action="store_true")
13
- p.add_argument("--revision", default="local-checkpoint")
14
- p.add_argument("--input", required=True)
15
- p.add_argument("--output", required=True)
16
- p.add_argument("--batch-size", type=int, default=8)
17
- p.add_argument("--max-length", type=int, default=4096)
18
- p.add_argument("--temperature", type=float, default=1.0)
19
- a = p.parse_args()
20
- assert a.temperature > 0
21
- torch.cuda.set_device(0)
22
- model, tokenizer = DecisionModel.from_base(a.model, a.revision) if a.base else DecisionModel.from_checkpoint(a.model)
23
- model = model.cuda().eval()
24
- rows = [json.loads(line) for line in Path(a.input).read_text().splitlines() if line.strip()]
25
- encoded = [encode(row, tokenizer, a.max_length) for row in rows]
26
- assert len({row["id"] for row in rows}) == len(rows)
27
- output = Path(a.output); output.parent.mkdir(parents=True, exist_ok=True)
28
- if output.exists(): raise RuntimeError("Refusing to overwrite predictions")
29
- with torch.inference_mode(), output.open("w") as f:
30
- for start in range(0, len(encoded), a.batch_size):
31
- examples = encoded[start:start + a.batch_size]
32
- batch = {key: value.cuda() if torch.is_tensor(value) else value for key, value in collate(examples, tokenizer.pad_token_id if tokenizer.pad_token_id is not None else tokenizer.eos_token_id).items()}
33
- torch.cuda.synchronize(); tick = time.perf_counter()
34
- with torch.autocast("cuda", dtype=torch.bfloat16): logits = model(**batch)
35
- torch.cuda.synchronize(); elapsed = time.perf_counter() - tick
36
- probabilities = (logits / a.temperature).softmax(-1).cpu().tolist(); scores = logits.cpu().tolist()
37
- for row, example, prob, score in zip(rows[start:start + a.batch_size], examples, probabilities, scores):
38
- k = example["nopts"]; pred = max(range(k), key=lambda i: prob[i])
39
- rec = {"id": row["id"], "family": row.get("family"), "label": row.get("label"), "prediction": pred, "prediction_key": row["options"][pred]["key"], "probabilities": prob[:k], "logits": score[:k], "temperature": a.temperature, "input_tokens": len(example["ids"]), "prompt_sha256": example["prompt_sha256"], "batch_elapsed_seconds": elapsed, "batch_size": len(examples)}
40
- if "score_values" in row: rec["expected_score"] = sum(value * probability for value, probability in zip(row["score_values"], prob[:k]))
41
- f.write(json.dumps(rec, ensure_ascii=False) + "\n")
42
- f.flush(); print(json.dumps({"completed": min(start + a.batch_size, len(rows)), "total": len(rows)}), flush=True)
43
- Path(str(output) + ".metadata.json").write_text(json.dumps({"model": a.model, "revision": a.revision, "temperature": a.temperature, "input_sha256": hashlib.sha256(Path(a.input).read_bytes()).hexdigest(), "predictions_sha256": hashlib.sha256(output.read_bytes()).hexdigest(), "rows": len(rows), "model_metadata": model.metadata}, indent=2))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
code/profile_guard.py DELETED
@@ -1,82 +0,0 @@
1
- """Fail-closed official FLA strict-config setup for one isolated process.
2
-
3
- No model/prompt/head changes. The guard prevents FLA STRICT's ordinary missing
4
- key fallback and records actual configured calls. This is a diagnostic module,
5
- not an installed change to the published wrapper or dependency environment.
6
- """
7
- import hashlib,importlib,json,os,sys
8
- from pathlib import Path
9
-
10
- def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
11
- def serialized(key):return json.dumps(key,separators=(',',':'),sort_keys=True)
12
- def config_fields(config):
13
- if isinstance(config,dict):return {k:config.get(k) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
14
- return {k:getattr(config,k,None) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
15
- def validate_profile(path,expected_sha):
16
- path=Path(path).resolve()
17
- if sha(path)!=expected_sha:raise ValueError('Profile hash changed')
18
- profile=json.loads(path.read_text())
19
- if profile['format']!='decision-fla-l2norm-profile-v1' or profile.get('model_family')!='Qwen/Qwen3.5-4B' or profile['cache_mode']!='strict':raise ValueError('Profile format/mode unsupported')
20
- if len(profile['files'])!=1 or profile['files'][0]['file']!='l2norm_fwd_kernel.json':raise ValueError('Unexpected profile file set')
21
- f=path.parent/'l2norm_fwd_kernel.json'
22
- if sha(f)!=profile['files'][0]['sha256']:raise ValueError('Explicit kernel config changed')
23
- data=json.loads(f.read_text());entries={}
24
- if data.get('default_config') is not None:raise ValueError('Implicit fallback defaults forbidden')
25
- for h,item in data['autotune_entries'].items():
26
- key=item['autotune_key'];encoded=serialized(key)
27
- if hashlib.md5(encoded.encode()).hexdigest()!=h or encoded in entries:raise ValueError('Invalid/duplicate key')
28
- if len(key)!=5 or key[0]!=128 or type(key[1]) is not int or not 1<=key[1]<=64 or key[2:]!=['torch.bfloat16','torch.bfloat16','torch.float32']:raise ValueError('Unsupported numerical key')
29
- c=item['config']
30
- if c['kwargs'].keys()!={'BT'} or c['kwargs']['BT'] not in [8,16,32,64] or c['num_warps'] not in [1,2,4,8,16] or c['num_stages']!=3 or c['num_ctas']!=1 or any(c.get(x) is not None for x in ['maxnreg','pre_hook','ir_override']):raise ValueError('Unexpected launch configuration')
31
- entries[encoded]=c
32
- if {json.loads(k)[1] for k in entries}!=set(range(1,65)):raise ValueError('Incomplete legal NB coverage')
33
- return path,profile,entries
34
-
35
- def attach_guard(kernel,cache_module,entries,telemetry):
36
- original=kernel.run
37
- def guarded(*args,**kwargs):
38
- if cache_module.FLA_CACHE_MODE is not cache_module.FlaCacheMode.STRICT:raise RuntimeError('FLA strict mode changed')
39
- key=cache_module.AutotuneKey.build(kernel.arg_names,kernel.keys,args,kwargs);encoded=serialized(list(key.autotune_key))
40
- if encoded not in entries:raise RuntimeError('Uncontracted FLA l2norm key: '+encoded)
41
- expected=entries[encoded];loaded=cache_module.load_cached_config(kernel.kernel_name,key)
42
- if config_fields(loaded)!=config_fields(expected):raise RuntimeError('FLA exact config lookup mismatch')
43
- if key.autotune_key in kernel.cache and config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('A conflicting in-process kernel cache exists')
44
- # Explicit official configuration load guarantees that the following original
45
- # run finds this exact cache entry and cannot perform timing-based autotune.
46
- kernel.maybe_load_cached_config(key)
47
- if key.autotune_key not in kernel.cache or config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Official strict config did not load')
48
- result=original(*args,**kwargs)
49
- if config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Kernel config changed during call')
50
- telemetry['calls']+=1;telemetry['keys'][encoded]=telemetry['keys'].get(encoded,0)+1
51
- return result
52
- kernel.run=guarded
53
- return original
54
-
55
- def install(profile_path,expected_sha):
56
- path,profile,entries=validate_profile(profile_path,expected_sha)
57
- if any(n=='fla' or n.startswith('fla.') for n in sys.modules):raise RuntimeError('Install profile before importing FLA; use a fresh isolated process')
58
- for name,wanted in {'FLA_CACHE_MODE':'strict','FLA_CONFIG_DIR':str(path.parent)}.items():
59
- actual=os.environ.get(name)
60
- if actual is not None and actual!=wanted:raise RuntimeError('Conflicting '+name)
61
- os.environ[name]=wanted
62
- import torch,triton,fla
63
- actual={'torch':str(torch.__version__),'hip':torch.version.hip,'triton':triton.__version__,'fla':fla.__version__}
64
- for name,value in actual.items():
65
- if value!=profile['runtime'][name]:raise RuntimeError('Runtime mismatch: '+name)
66
- if not torch.cuda.is_available():raise RuntimeError('Profile is only qualified for the specified ROCm GPU')
67
- arch=torch.cuda.get_device_properties(0).gcnArchName.split(':')[0]
68
- if arch!=profile['runtime']['gpu_arch']:raise RuntimeError('Unsupported GPU architecture '+arch)
69
- module=importlib.import_module('fla.modules.l2norm');cache_module=importlib.import_module('fla.ops.utils.cache');root=Path(fla.__file__).parent
70
- for name,value in profile['fla_source_sha256'].items():
71
- if sha(root/name)!=value:raise RuntimeError('Pinned FLA source changed: '+name)
72
- kernel=module.l2norm_fwd_kernel
73
- if kernel.kernel_name!='l2norm_fwd_kernel' or kernel.keys!=['D','NB'] or kernel.cache:raise RuntimeError('Kernel identity or fresh-cache precondition failed')
74
- telemetry={'profile_sha256':expected_sha,'status':'installed','calls':0,'keys':{},'strict_guard':True,'unknown_keys':'raise','autotune_fallback_permitted':False,'runtime':actual,'gpu_arch':arch,'process_scope':'one explicitly profiled Nox model; other model loading in this process is not supported'}
75
- attach_guard(kernel,cache_module,entries,telemetry)
76
- # The validated inference path only uses the vectorized D128 forward kernel.
77
- # Other dimensions/backward must not silently enter a different autotuner.
78
- for name in ['l2norm_fwd_kernel1','l2norm_bwd_kernel','l2norm_bwd_kernel1']:
79
- other=getattr(module,name)
80
- def reject(*args,_name=name,**kwargs):raise RuntimeError('Uncontracted normalization kernel: '+_name)
81
- other.run=reject
82
- return telemetry
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
code/runtime_profile.py DELETED
@@ -1,36 +0,0 @@
1
- """Bundle-local automatic launch-profile binding, before importing FLA.
2
-
3
- One profile is active per Python process. Repeated loading of the same verified
4
- profile is allowed; mixing with an unprofiled/different-profile model is not.
5
- """
6
- import hashlib,importlib.util,json,os,sys,types
7
- from pathlib import Path
8
- STATE='_decision_process_normalization_profile_v1'
9
- def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
10
- def safe(root,relative):
11
- p=Path(relative)
12
- if p.is_absolute() or '..' in p.parts:raise ValueError('Unsafe runtime-profile path')
13
- return root/p
14
-
15
- def ensure_profile(bundle):
16
- bundle=Path(bundle).resolve();runtime=json.loads((bundle/'runtime.json').read_text());spec=runtime.get('normalization_profile');active=sys.modules.get(STATE)
17
- if spec is None:
18
- if active is not None:raise RuntimeError('Load unprofiled and profiled Decision models in separate processes')
19
- return None
20
- if spec.get('kind')!='decision-fla-l2norm-profile-v1' or spec.get('validated_arch')!='gfx942':raise ValueError('Unknown normalization profile contract')
21
- profile=safe(bundle,spec['profile_file']);guard=safe(bundle,spec['guard_file'])
22
- if sha(profile)!=spec['profile_sha256'] or sha(guard)!=spec['guard_sha256']:raise ValueError('Bound runtime profile bytes changed')
23
- if json.loads((bundle/'decision_config.json').read_text()).get('base_model')!='Qwen/Qwen3.5-4B':raise ValueError('This bundle profile is bound to the validated Nox family only')
24
- if active is not None:
25
- if active.profile_sha256!=spec['profile_sha256'] or active.guard_sha256!=spec['guard_sha256']:raise RuntimeError('Different Decision normalization profile is already active; use a separate process')
26
- active.guard.validate_profile(profile,spec['profile_sha256'])
27
- if os.environ.get('FLA_CACHE_MODE')!='strict' or os.environ.get('FLA_CONFIG_DIR')!=active.profile_dir:raise RuntimeError('Active FLA profile environment changed')
28
- return {'profile_sha256':active.profile_sha256,'guard_sha256':active.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
29
- module_spec=importlib.util.spec_from_file_location('decision_profile_guard_'+spec['guard_sha256'][:16],guard);module=importlib.util.module_from_spec(module_spec);module_spec.loader.exec_module(module)
30
- telemetry=module.install(profile,spec['profile_sha256'])
31
- state=types.ModuleType(STATE);state.profile_sha256=spec['profile_sha256'];state.guard_sha256=spec['guard_sha256'];state.profile_dir=str(profile.parent);state.guard=module;state.telemetry=telemetry;sys.modules[STATE]=state
32
- return {'profile_sha256':state.profile_sha256,'guard_sha256':state.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
33
-
34
- def active_telemetry():
35
- state=sys.modules.get(STATE)
36
- return None if state is None else dict(state.telemetry)