Publish clean Decision model repository
Browse filesSigned-off-by: Xunzhuo <[email protected]>
This view is limited to 50 files because it contains too many changes. See raw diff
- Dockerfile.runtime +0 -12
- MATERIALS.json +0 -28
- NORMALIZATION_RUNTIME.md +0 -5
- NULL_DESCRIPTION_RENDERING.json +0 -7
- README.md +14 -52
- RUNTIME-RELEASE.json +0 -11
- RUNTIME.md +0 -52
- RUNTIME_BINDING.json +0 -76
- SERVING_OPTIMIZATION.json +0 -34
- SOURCE_BUNDLE_MANIFEST.json +0 -173
- USAGE.md +0 -59
- WEIGHTING.md +0 -21
- assets/decision-expanded-old_core-600px.png +0 -3
- assets/decision-expanded-old_core.pdf +0 -0
- assets/decision-expanded-old_core.png +0 -3
- assets/decision-expanded-old_core.svg +0 -1620
- assets/decision-expanded-overview-600px.png +0 -0
- assets/decision-expanded-overview.pdf +0 -0
- assets/decision-expanded-overview.png +0 -3
- assets/decision-expanded-overview.svg +0 -804
- assets/decision-expanded-ranking-600px.png +0 -0
- assets/decision-expanded-ranking.pdf +0 -0
- assets/decision-expanded-ranking.png +0 -3
- assets/decision-expanded-ranking.svg +0 -501
- assets/decision-expanded-v3_core-600px.png +0 -3
- assets/decision-expanded-v3_core.pdf +0 -0
- assets/decision-expanded-v3_core.png +0 -3
- assets/decision-expanded-v3_core.svg +0 -1621
- assets/decision-expanded-v4-600px.png +0 -0
- assets/decision-expanded-v4.pdf +0 -0
- assets/decision-expanded-v4.png +0 -3
- assets/decision-expanded-v4.svg +0 -668
- assets/decision-expanded-v5-600px.png +0 -0
- assets/decision-expanded-v5.pdf +0 -0
- assets/decision-expanded-v5.png +0 -3
- assets/decision-expanded-v5.svg +0 -804
- assets/decision-family-header.png +0 -3
- assets/decision-matrix.pdf +0 -0
- assets/decision-matrix.svg +0 -1338
- assets/decision-question-scaling-600px.png +0 -0
- assets/decision-question-scaling.pdf +0 -0
- assets/decision-question-scaling.svg +0 -239
- assets/decision-ranking.pdf +0 -0
- assets/decision-ranking.svg +0 -369
- bundle-manifest.json +0 -243
- code/decision_api.py +0 -198
- code/decision_model.py +0 -173
- code/predict.py +0 -43
- code/profile_guard.py +0 -82
- code/runtime_profile.py +0 -36
Dockerfile.runtime
DELETED
|
@@ -1,12 +0,0 @@
|
|
| 1 |
-
# Public base verified by manifest digest and critical PyTorch file hashes.
|
| 2 |
-
# CPU build/import validation is separate from model GPU qualification; see RUNTIME.md.
|
| 3 |
-
FROM vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339
|
| 4 |
-
COPY runtime-fla-requirements.lock /tmp/runtime-fla-requirements.lock
|
| 5 |
-
RUN python3 -m pip install --no-cache-dir --no-index --no-deps --require-hashes --target /opt/decision-fla -r /tmp/runtime-fla-requirements.lock
|
| 6 |
-
ENV PYTHONPATH=/opt/decision-fla
|
| 7 |
-
COPY pyproject.toml /opt/decision-wrapper/pyproject.toml
|
| 8 |
-
COPY src/ /opt/decision-wrapper/src/
|
| 9 |
-
RUN python3 -m pip install --no-cache-dir --no-deps --no-build-isolation /opt/decision-wrapper
|
| 10 |
-
WORKDIR /model
|
| 11 |
-
ENTRYPOINT []
|
| 12 |
-
CMD ["/bin/bash"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
MATERIALS.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"kind": "latest-public-comparison-documents",
|
| 3 |
-
"public_models": 15,
|
| 4 |
-
"scored_questions": 3766,
|
| 5 |
-
"tasks": 54,
|
| 6 |
-
"source_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
|
| 7 |
-
"files": {
|
| 8 |
-
".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
|
| 9 |
-
"DIAGNOSTICS.md": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa",
|
| 10 |
-
"EVALUATION.md": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a",
|
| 11 |
-
"QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
|
| 12 |
-
"README.md": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
|
| 13 |
-
"SENSITIVITY.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
|
| 14 |
-
"TASKS.md": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247",
|
| 15 |
-
"USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
|
| 16 |
-
"WEIGHTING.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
|
| 17 |
-
"assets/decision-matrix.pdf": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b",
|
| 18 |
-
"assets/decision-matrix.png": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696",
|
| 19 |
-
"assets/decision-matrix.svg": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc",
|
| 20 |
-
"assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
|
| 21 |
-
"assets/decision-ranking.pdf": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f",
|
| 22 |
-
"assets/decision-ranking.png": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9",
|
| 23 |
-
"assets/decision-ranking.svg": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e",
|
| 24 |
-
"metrics/benchmark.json": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c",
|
| 25 |
-
"metrics/evaluation-provenance.json": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9",
|
| 26 |
-
"metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
|
| 27 |
-
}
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
NORMALIZATION_RUNTIME.md
DELETED
|
@@ -1,5 +0,0 @@
|
|
| 1 |
-
# Validated normalization runtime
|
| 2 |
-
|
| 3 |
-
This Nox bundle installs its recorded FLA normalization profile through the default public entrypoint. It requires the pinned ROCm runtime on gfx942 and a fresh process. The profile covers dimension 128, BF16 input/output, FP32 reciprocal norms, and buckets 1–64 for the actual 32 normalized value heads, batch size at most 8 and complete inputs at most 16,384 tokens. Unknown keys fail closed. Weights, tokenizer, prompt, readout and temperature remain unchanged from the source export.
|
| 4 |
-
|
| 5 |
-
No private compilation cache is required. Other FLA kernels retain normal runtime behavior. Numerical validation of this newly bound bundle is still required. Earlier latency measurements do not measure this runtime or candidate.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
NULL_DESCRIPTION_RENDERING.json
DELETED
|
@@ -1,7 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"rule": "For Choice only, replace a null description with its original key text before tokenization.",
|
| 3 |
-
"preserved": "Non-null descriptions including empty string; keys/order; Noul and Score; original request; weights/tokenizer/temperature/numeric runtime.",
|
| 4 |
-
"source_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 5 |
-
"qualification": "Candidate only; full fixed benchmark and independent public loading proof required.",
|
| 6 |
-
"not_a_weight_training_update": true
|
| 7 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
README.md
CHANGED
|
@@ -7,7 +7,6 @@ tags:
|
|
| 7 |
- decision-model
|
| 8 |
- classification
|
| 9 |
- qwen3_5
|
| 10 |
-
- custom-code
|
| 11 |
- pytorch
|
| 12 |
- rocm
|
| 13 |
---
|
|
@@ -58,75 +57,38 @@ Accuracy (%). Overall weights: Decisions **30%**, Composition **25%**, Reading *
|
|
| 58 |
|
| 59 |

|
| 60 |
|
| 61 |
-
[All 54 tasks](TASKS.md) · [Order, missing-evidence and calibration diagnostics](DIAGNOSTICS.md) · [Methods and uncertainty](EVALUATION.md)
|
| 62 |
|
| 63 |
## More questions, one request
|
| 64 |
|
| 65 |

|
| 66 |
|
| 67 |
-
Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](QUESTION-SCALING.md).
|
| 68 |
|
| 69 |
-
##
|
| 70 |
-
|
| 71 |
-
Use the [official TypeSafe Python SDK](https://docs.typesafe.ai/sdk/python/usage) with your SystemOne-compatible endpoint, configured to serve `Decision-1.0-Nox-4B`. Replace the example URL and API key with your own.
|
| 72 |
|
| 73 |
```bash
|
| 74 |
-
|
| 75 |
```
|
| 76 |
|
| 77 |
-
``
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
with TypeSafeClient(
|
| 81 |
-
api_key="YOUR_ENDPOINT_API_KEY",
|
| 82 |
-
base_url="https://your-decision-endpoint.example",
|
| 83 |
-
model="Decision-1.0-Nox-4B",
|
| 84 |
-
) as client:
|
| 85 |
-
result = client.system_one(
|
| 86 |
-
state="Customer reports a duplicate charge and asks for a refund.",
|
| 87 |
-
questions={
|
| 88 |
-
"route": Choice(
|
| 89 |
-
instructions="Which team should handle this request?",
|
| 90 |
-
criteria={"billing": "Payments and refunds", "technical": "Product faults"},
|
| 91 |
-
),
|
| 92 |
-
"refund_requested": Noul(instructions="Did the customer request a refund?"),
|
| 93 |
-
},
|
| 94 |
-
)
|
| 95 |
-
print(result.choices["route"].choice)
|
| 96 |
-
print(result.nouls["refund_requested"].noul)
|
| 97 |
-
```
|
| 98 |
|
| 99 |
-
The
|
|
|
|
|
|
|
| 100 |
|
| 101 |
```bash
|
| 102 |
curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
|
| 103 |
-H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
|
| 104 |
-H 'Content-Type: application/json' \
|
| 105 |
-
--data-raw '{
|
| 106 |
-
"model": "Decision-1.0-Nox-4B",
|
| 107 |
-
"state": "Customer reports a duplicate charge and asks for a refund.",
|
| 108 |
-
"questions": {
|
| 109 |
-
"route": {
|
| 110 |
-
"type": "choice",
|
| 111 |
-
"instructions": "Which team should handle this request?",
|
| 112 |
-
"criteria": {
|
| 113 |
-
"billing": "Payments and refunds",
|
| 114 |
-
"technical": "Product faults"
|
| 115 |
-
}
|
| 116 |
-
},
|
| 117 |
-
"refund_requested": {
|
| 118 |
-
"type": "noul",
|
| 119 |
-
"instructions": "Did the customer request a refund?"
|
| 120 |
-
}
|
| 121 |
-
}
|
| 122 |
-
}'
|
| 123 |
```
|
| 124 |
|
| 125 |
-
|
| 126 |
-
|
| 127 |
-
Choice candidates with a null description use their ID text, which may increase input tokens.
|
| 128 |
|
| 129 |
-
The complete state, question and candidates
|
| 130 |
|
| 131 |
## Architecture
|
| 132 |
|
|
@@ -134,6 +96,6 @@ The complete state, question and candidates must fit 16,384 tokens; overflow is
|
|
| 134 |
|
| 135 |
A causal Qwen3.5 text backbone combines gated linear and full attention. A shared candidate head reads candidate endpoints and the final query vector. Each question uses one forward pass; questions run independently in batches of eight.
|
| 136 |
|
| 137 |
-
[Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg)
|
| 138 |
|
| 139 |
Adapted from [Qwen3.5-4B](https://huggingface.co/Qwen/Qwen3.5-4B). It evaluates supplied evidence without live retrieval; confidence does not guarantee correctness. [License](LICENSE) · [Attributions](ATTRIBUTIONS.md).
|
|
|
|
| 7 |
- decision-model
|
| 8 |
- classification
|
| 9 |
- qwen3_5
|
|
|
|
| 10 |
- pytorch
|
| 11 |
- rocm
|
| 12 |
---
|
|
|
|
| 57 |
|
| 58 |

|
| 59 |
|
| 60 |
+
[All 54 tasks](evaluation/TASKS.md) · [Order, missing-evidence and calibration diagnostics](evaluation/DIAGNOSTICS.md) · [Methods and uncertainty](evaluation/EVALUATION.md)
|
| 61 |
|
| 62 |
## More questions, one request
|
| 63 |
|
| 64 |

|
| 65 |
|
| 66 |
+
Distinct Choice questions at a fixed **499 input tokens per question**. Thirty measurements per point across six independently loaded processes on an otherwise idle AMD gfx942 GPU. Python latency includes tokenization and inference; loading and network are excluded. These measurements precede null-description normalization and use explicit descriptions. [p50, p95 and measurement scope](evaluation/QUESTION-SCALING.md).
|
| 67 |
|
| 68 |
+
## Download the complete model repository
|
|
|
|
|
|
|
| 69 |
|
| 70 |
```bash
|
| 71 |
+
hf download llm-semantic-router/Decision-1.0-Nox-4B --local-dir Decision-1.0-Nox-4B
|
| 72 |
```
|
| 73 |
|
| 74 |
+
This downloads the complete model release. The root `config.json` lists the backbone, tokenizer, decision head, and calibration files.
|
| 75 |
+
|
| 76 |
+
## Serve with vLLM Semantic Router
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 77 |
|
| 78 |
+
This repository contains model data only. Use the vLLM Semantic Router Decision runtime to load `llm-semantic-router/Decision-1.0-Nox-4B` and serve Choice, Noul, and Score requests. The serving implementation and its dependencies live in vLLM Semantic Router; this release does not bundle executable model code. `transformers.AutoModel.from_pretrained` cannot load the custom Decision head directly.
|
| 79 |
+
|
| 80 |
+
After configuring a compatible Decision endpoint, send a [SystemOne request](https://docs.typesafe.ai/api) (replace the placeholder URL and key):
|
| 81 |
|
| 82 |
```bash
|
| 83 |
curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
|
| 84 |
-H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
|
| 85 |
-H 'Content-Type: application/json' \
|
| 86 |
+
--data-raw '{"model":"Decision-1.0-Nox-4B","state":"Customer requests a refund.","questions":{"route":{"type":"choice","instructions":"Which team should handle this?","criteria":{"billing":"Payments and refunds","technical":"Product faults"}}}}'
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 87 |
```
|
| 88 |
|
| 89 |
+
The Hugging Face repository is a model download, not a hosted inference endpoint.
|
|
|
|
|
|
|
| 90 |
|
| 91 |
+
The published model's complete state, question, and candidates have a 16,384-token input limit. See the [evaluation scope](evaluation/EVALUATION.md) for measured conditions.
|
| 92 |
|
| 93 |
## Architecture
|
| 94 |
|
|
|
|
| 96 |
|
| 97 |
A causal Qwen3.5 text backbone combines gated linear and full attention. A shared candidate head reads candidate endpoints and the final query vector. Each question uses one forward pass; questions run independently in batches of eight.
|
| 98 |
|
| 99 |
+
[Candidate head](assets/readout.png) · [Vector architecture](assets/architecture.svg)
|
| 100 |
|
| 101 |
Adapted from [Qwen3.5-4B](https://huggingface.co/Qwen/Qwen3.5-4B). It evaluates supplied evidence without live retrieval; confidence does not guarantee correctness. [License](LICENSE) · [Attributions](ATTRIBUTIONS.md).
|
RUNTIME-RELEASE.json
DELETED
|
@@ -1,11 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"release": "v1.3.2",
|
| 3 |
-
"kind": "SystemOne Choice null-description semantics",
|
| 4 |
-
"rule": "A null Choice description uses the original candidate key as its description.",
|
| 5 |
-
"weights_tokenizer_temperature_unchanged": true,
|
| 6 |
-
"explicit_descriptions_and_Noul_Score_unchanged": true,
|
| 7 |
-
"bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 8 |
-
"qualified_statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
| 9 |
-
"exact_public_contract_proof_sha256": "ebf27c7aa2bb4a12ba2cbb27e4096e46cb43ac4a20b9d3e1bc34f62777a1e63a",
|
| 10 |
-
"runtime_note": "This is an API rendering improvement, not a newly trained checkpoint."
|
| 11 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
RUNTIME.md
DELETED
|
@@ -1,52 +0,0 @@
|
|
| 1 |
-
# Public ROCm runtime
|
| 2 |
-
|
| 3 |
-
The package has a public, digest-pinned installation path. `Dockerfile.runtime` starts from `vllm/vllm-openai-rocm@sha256:1fd21abe66455b4df5a2e83629e97cdcc9d58913b16052d8118b92b239792339`, adds two hash-checked FLA wheels, and installs this repository's loading wrapper. It keeps the base image's ROCm PyTorch and Triton builds. The vLLM server is not used by Decision inference.
|
| 4 |
-
|
| 5 |
-
**Validation boundary:** the public registry manifest, base-image ancestry, package metadata and critical PyTorch binary hashes have been checked. The recipe built successfully and passed CPU imports and real AMD ROCm gfx942 GPU GPU inference for both released bundles. On the packaged three-question Choice/Noul/Score example, its complete responses matched the qualified research runtime exactly; the wrapper matched the direct engine and rejected an oversized complete input. Evidence is in `runtime-build-provenance.json`. This example establishes a working public installation path; it is not a full rerun of the quality or timing benchmark. Published benchmark results use the qualified runtime in each bundle's `runtime.json`.
|
| 6 |
-
|
| 7 |
-
## Build and run
|
| 8 |
-
|
| 9 |
-
Use an AMD ROCm-compatible Linux host with Docker and the required GPU driver. Check the [AMD PyTorch installation and host prerequisites](https://rocm.docs.amd.com/projects/ai-ecosystem/en/latest/frameworks/pytorch/install.html). CPU and MPS inference are not supported by this model engine; no NVIDIA validation is claimed.
|
| 10 |
-
|
| 11 |
-
Run these commands from the downloaded model repository containing `Dockerfile.runtime`, `runtime-fla-requirements.lock`, `pyproject.toml`, `src/`, and the model bundle:
|
| 12 |
-
|
| 13 |
-
```bash
|
| 14 |
-
docker build --pull -f Dockerfile.runtime -t decision-runtime:1.0 .
|
| 15 |
-
mkdir -p runtime-output
|
| 16 |
-
docker run --rm \
|
| 17 |
-
--device=/dev/kfd --device=/dev/dri --group-add video --ipc=host \
|
| 18 |
-
-v "$PWD":/model:ro -v "$PWD/runtime-output":/output \
|
| 19 |
-
decision-runtime:1.0 \
|
| 20 |
-
python3 -m decision.example /model --local-files-only --output /output/example.json
|
| 21 |
-
```
|
| 22 |
-
|
| 23 |
-
This loads locally without a Hub token. The example tests Choice, Noul and Score through the wrapper and compares them with the frozen direct engine; inspect its actual result rather than assuming a predicted answer. Keep the runtime checks enabled. If they report a mismatch, resolve the cause before using this environment to reproduce benchmark claims. The explicit `allow_unvalidated_runtime=True` option is for separately labeled experiments, not benchmark reproduction.
|
| 24 |
-
|
| 25 |
-
The image is large because its public base includes the vLLM development environment. No Decision weights, dataset, user credentials or private runtime image are required to build it. Model weights are mounted at execution time.
|
| 26 |
-
|
| 27 |
-
## What is pinned
|
| 28 |
-
|
| 29 |
-
The public base locks the existing OS, ROCm libraries and Python dependency environment by content digest. Its exact observed core versions are:
|
| 30 |
-
|
| 31 |
-
| Component | Observed value |
|
| 32 |
-
|---|---|
|
| 33 |
-
| Python | 3.12.13 |
|
| 34 |
-
| PyTorch | 2.12.0+git6bbd260 |
|
| 35 |
-
| PyTorch commit | 6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5 |
|
| 36 |
-
| ROCm userspace / HIP build | 7.2.3 / 7.2.53211 |
|
| 37 |
-
| Triton distribution / imported version | 3.7.1+gitf0b55c07 / 3.7.1 |
|
| 38 |
-
| Transformers | 5.17.0 |
|
| 39 |
-
| Tokenizers / Safetensors | 0.23.2 / 0.8.0 |
|
| 40 |
-
| NumPy / Einops | 2.3.5 / 0.8.2 |
|
| 41 |
-
| Hugging Face Hub | 1.31.0 |
|
| 42 |
-
| FLA core / Flash Linear Attention | 0.5.2 / 0.5.2 |
|
| 43 |
-
|
| 44 |
-
`runtime-provenance.json` records the live registry manifest response, 36 shared base layers, selected actual binary SHA256 values, observed versions, wheel URLs and wheel hashes. The FLA wheels were downloaded and their SHA256 values matched the qualified overlay. `runtime-fla-requirements.lock` intentionally covers only that overlay; it is not a standalone dependency lock for an arbitrary system.
|
| 45 |
-
|
| 46 |
-
The official [FLA installation guide](https://github.com/fla-org/flash-linear-attention/blob/main/INSTALL.md) separates backend PyTorch installation from FLA and documents `--no-deps` for pre-release/custom Torch builds. This recipe uses that boundary and never asks pip to replace Torch or Triton. Transformers is supplied by the pinned base; see its [official installation documentation](https://huggingface.co/docs/transformers/installation) for the general package installation model.
|
| 47 |
-
|
| 48 |
-
## Portability boundary
|
| 49 |
-
|
| 50 |
-
The public [PyTorch ROCm 7.2 wheel index](https://download.pytorch.org/whl/rocm7.2/torch/) contains ordinary release wheels such as `2.12.0+rocm7.2`. They are a different artifact from the qualified `2.12.0+git6bbd260` build and are not interchangeable evidence. The latter's [source commit is public](https://github.com/pytorch/pytorch/commit/6bbd26020da1c6dc198625dfcdd968b1e4e6b1c5), but a source commit alone does not reproduce compiler flags, linked libraries and binary behavior.
|
| 51 |
-
|
| 52 |
-
An alternative runtime must recheck the installed versions, actual FLA Gated DeltaNet dispatch, BF16 backbone plus FP32 head, fixed prompt rendering, batch size eight, and model output agreement. Runtime changes can shift probabilities near a decision boundary even with identical weights. The supplied engine uses FLA Gated DeltaNet, reference PyTorch causal convolution and SDPA; selecting a different kernel is a new runtime configuration.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
RUNTIME_BINDING.json
DELETED
|
@@ -1,76 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"source_bundle_manifest_sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4",
|
| 3 |
-
"profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
|
| 4 |
-
"profile_validation_receipt_sha256": "7a66474325b5575fdd15a33e5e0340c19e809bdcc22adf4cbb45baa0e1b32693",
|
| 5 |
-
"unchanged_files": [
|
| 6 |
-
{
|
| 7 |
-
"file": "backbone/config.json",
|
| 8 |
-
"bytes": 1978,
|
| 9 |
-
"sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"file": "backbone/model-00001-of-00003.safetensors",
|
| 13 |
-
"bytes": 3991295368,
|
| 14 |
-
"sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
|
| 15 |
-
},
|
| 16 |
-
{
|
| 17 |
-
"file": "backbone/model-00002-of-00003.safetensors",
|
| 18 |
-
"bytes": 3979828128,
|
| 19 |
-
"sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
|
| 20 |
-
},
|
| 21 |
-
{
|
| 22 |
-
"file": "backbone/model-00003-of-00003.safetensors",
|
| 23 |
-
"bytes": 440425856,
|
| 24 |
-
"sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
|
| 25 |
-
},
|
| 26 |
-
{
|
| 27 |
-
"file": "backbone/model.safetensors.index.json",
|
| 28 |
-
"bytes": 33047,
|
| 29 |
-
"sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
|
| 30 |
-
},
|
| 31 |
-
{
|
| 32 |
-
"file": "chat_template.jinja",
|
| 33 |
-
"bytes": 7756,
|
| 34 |
-
"sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
|
| 35 |
-
},
|
| 36 |
-
{
|
| 37 |
-
"file": "code/decision_model.py",
|
| 38 |
-
"bytes": 10114,
|
| 39 |
-
"sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
|
| 40 |
-
},
|
| 41 |
-
{
|
| 42 |
-
"file": "code/predict.py",
|
| 43 |
-
"bytes": 3164,
|
| 44 |
-
"sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
|
| 45 |
-
},
|
| 46 |
-
{
|
| 47 |
-
"file": "decision_config.json",
|
| 48 |
-
"bytes": 753,
|
| 49 |
-
"sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
|
| 50 |
-
},
|
| 51 |
-
{
|
| 52 |
-
"file": "decision_head.safetensors",
|
| 53 |
-
"bytes": 10529624,
|
| 54 |
-
"sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
|
| 55 |
-
},
|
| 56 |
-
{
|
| 57 |
-
"file": "temperature.json",
|
| 58 |
-
"bytes": 6083,
|
| 59 |
-
"sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
|
| 60 |
-
},
|
| 61 |
-
{
|
| 62 |
-
"file": "tokenizer.json",
|
| 63 |
-
"bytes": 19989325,
|
| 64 |
-
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
|
| 65 |
-
},
|
| 66 |
-
{
|
| 67 |
-
"file": "tokenizer_config.json",
|
| 68 |
-
"bytes": 1123,
|
| 69 |
-
"sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
|
| 70 |
-
}
|
| 71 |
-
],
|
| 72 |
-
"training_selection_calibration_unchanged": true,
|
| 73 |
-
"recomputed_or_recalibrated": false,
|
| 74 |
-
"private_compiled_cache_required": false,
|
| 75 |
-
"new_bundle_offline_proof_required": true
|
| 76 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
SERVING_OPTIMIZATION.json
DELETED
|
@@ -1,34 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"format": "joint-serving-runtime-candidate-v1",
|
| 3 |
-
"family": "Nox",
|
| 4 |
-
"source_publication_revision": "ad089ad3a5dc9a7a21e6d96db546bb53e2212654",
|
| 5 |
-
"source_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
|
| 6 |
-
"source_release_manifest_sha256": "3e11cc6011860ea245c53eb44b9da61a85d5feb5c6caa5da95428d63ede6d712",
|
| 7 |
-
"source_api_sha256": "1b068eccdffd3c3b67bfa52f8f526e6b482d92551c927668ad28a794767f8a40",
|
| 8 |
-
"candidate_api_sha256": "273f6f10f22d5a68b8db34cfcbd35407fb43d8030f6d7cd188bcf118dd90a152",
|
| 9 |
-
"changed_inference_files": [
|
| 10 |
-
"code/decision_api.py"
|
| 11 |
-
],
|
| 12 |
-
"held_fixed": [
|
| 13 |
-
"weights",
|
| 14 |
-
"tokenizer",
|
| 15 |
-
"prompt",
|
| 16 |
-
"temperature",
|
| 17 |
-
"normalization_profile",
|
| 18 |
-
"BF16_backbone_FP32_head",
|
| 19 |
-
"batch8",
|
| 20 |
-
"input_limit16384",
|
| 21 |
-
"public_wrapper"
|
| 22 |
-
],
|
| 23 |
-
"shared_state_neural_cache": false,
|
| 24 |
-
"cross_request_cache": false,
|
| 25 |
-
"timing_evidence": {
|
| 26 |
-
"analysis/decoder-joint-serving-v1/COMPLETED-PARITY.json": "4187fb76432eb69b263cb6d5ad55f10a09aef384fe405f3ad204acf1edc761b5",
|
| 27 |
-
"analysis/decoder-joint-timing-v1/INDEPENDENTLY-REVIEWED.json": "1b416c8556c2305ffd18bd1e3cd523964820a5e2a3029cf447f20b2829ba82f7",
|
| 28 |
-
"results/joint-serving-timing-v1/summary/SUMMARY.json": "501f25efd1813b566d77e98278229440b7a3b4a2358def1462c0171a33e9f3be",
|
| 29 |
-
"analysis/decoder4b/joint-timing-actual-review-v1/REVIEW.json": "e94fec9d9c6d487200ef7ea1cf82319b33c4836c08d83f0b856bf8590f2fbee6"
|
| 30 |
-
},
|
| 31 |
-
"candidate_default_entrypoint_proof_required": true,
|
| 32 |
-
"published": false,
|
| 33 |
-
"adoption_authorized": false
|
| 34 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
SOURCE_BUNDLE_MANIFEST.json
DELETED
|
@@ -1,173 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"format": "research-pointer-bundle-v1",
|
| 3 |
-
"status": "candidate-export-awaiting-independent-reload-and-quality-gates",
|
| 4 |
-
"files": [
|
| 5 |
-
{
|
| 6 |
-
"file": "backbone/config.json",
|
| 7 |
-
"bytes": 1978,
|
| 8 |
-
"sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
|
| 9 |
-
},
|
| 10 |
-
{
|
| 11 |
-
"file": "backbone/model-00001-of-00003.safetensors",
|
| 12 |
-
"bytes": 3991295368,
|
| 13 |
-
"sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
|
| 14 |
-
},
|
| 15 |
-
{
|
| 16 |
-
"file": "backbone/model-00002-of-00003.safetensors",
|
| 17 |
-
"bytes": 3979828128,
|
| 18 |
-
"sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
|
| 19 |
-
},
|
| 20 |
-
{
|
| 21 |
-
"file": "backbone/model-00003-of-00003.safetensors",
|
| 22 |
-
"bytes": 440425856,
|
| 23 |
-
"sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
|
| 24 |
-
},
|
| 25 |
-
{
|
| 26 |
-
"file": "backbone/model.safetensors.index.json",
|
| 27 |
-
"bytes": 33047,
|
| 28 |
-
"sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
|
| 29 |
-
},
|
| 30 |
-
{
|
| 31 |
-
"file": "chat_template.jinja",
|
| 32 |
-
"bytes": 7756,
|
| 33 |
-
"sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
|
| 34 |
-
},
|
| 35 |
-
{
|
| 36 |
-
"file": "code/decision_api.py",
|
| 37 |
-
"bytes": 6724,
|
| 38 |
-
"sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21"
|
| 39 |
-
},
|
| 40 |
-
{
|
| 41 |
-
"file": "code/decision_model.py",
|
| 42 |
-
"bytes": 10114,
|
| 43 |
-
"sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
|
| 44 |
-
},
|
| 45 |
-
{
|
| 46 |
-
"file": "code/predict.py",
|
| 47 |
-
"bytes": 3164,
|
| 48 |
-
"sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
|
| 49 |
-
},
|
| 50 |
-
{
|
| 51 |
-
"file": "decision_config.json",
|
| 52 |
-
"bytes": 753,
|
| 53 |
-
"sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
|
| 54 |
-
},
|
| 55 |
-
{
|
| 56 |
-
"file": "decision_head.safetensors",
|
| 57 |
-
"bytes": 10529624,
|
| 58 |
-
"sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
|
| 59 |
-
},
|
| 60 |
-
{
|
| 61 |
-
"file": "runtime.json",
|
| 62 |
-
"bytes": 378,
|
| 63 |
-
"sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9"
|
| 64 |
-
},
|
| 65 |
-
{
|
| 66 |
-
"file": "temperature.json",
|
| 67 |
-
"bytes": 6083,
|
| 68 |
-
"sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
|
| 69 |
-
},
|
| 70 |
-
{
|
| 71 |
-
"file": "tokenizer.json",
|
| 72 |
-
"bytes": 19989325,
|
| 73 |
-
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
|
| 74 |
-
},
|
| 75 |
-
{
|
| 76 |
-
"file": "tokenizer_config.json",
|
| 77 |
-
"bytes": 1123,
|
| 78 |
-
"sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
|
| 79 |
-
}
|
| 80 |
-
],
|
| 81 |
-
"tensors": [
|
| 82 |
-
{
|
| 83 |
-
"file": "backbone/model-00001-of-00003.safetensors",
|
| 84 |
-
"elements": 1995638528,
|
| 85 |
-
"elements_by_dtype": {
|
| 86 |
-
"BF16": 1995638528
|
| 87 |
-
}
|
| 88 |
-
},
|
| 89 |
-
{
|
| 90 |
-
"file": "backbone/model-00002-of-00003.safetensors",
|
| 91 |
-
"elements": 1989900928,
|
| 92 |
-
"elements_by_dtype": {
|
| 93 |
-
"BF16": 1989900928
|
| 94 |
-
}
|
| 95 |
-
},
|
| 96 |
-
{
|
| 97 |
-
"file": "backbone/model-00003-of-00003.safetensors",
|
| 98 |
-
"elements": 220211840,
|
| 99 |
-
"elements_by_dtype": {
|
| 100 |
-
"BF16": 220211840
|
| 101 |
-
}
|
| 102 |
-
},
|
| 103 |
-
{
|
| 104 |
-
"file": "decision_head.safetensors",
|
| 105 |
-
"elements": 2632192,
|
| 106 |
-
"elements_by_dtype": {
|
| 107 |
-
"F32": 2632192
|
| 108 |
-
}
|
| 109 |
-
}
|
| 110 |
-
],
|
| 111 |
-
"source_checkpoint_files": [
|
| 112 |
-
{
|
| 113 |
-
"file": "backbone/config.json",
|
| 114 |
-
"bytes": 1977,
|
| 115 |
-
"sha256": "a5ed4156fda05f0f9149c66964d6165916754e7355488a8a07d9b0398acdbdb9"
|
| 116 |
-
},
|
| 117 |
-
{
|
| 118 |
-
"file": "backbone/model-00001-of-00005.safetensors",
|
| 119 |
-
"bytes": 3992269864,
|
| 120 |
-
"sha256": "d65565b8988a20b07988db09ed2321111e57b4fe656112fe4636715f8418c40b"
|
| 121 |
-
},
|
| 122 |
-
{
|
| 123 |
-
"file": "backbone/model-00002-of-00005.safetensors",
|
| 124 |
-
"bytes": 3990302320,
|
| 125 |
-
"sha256": "99197c2df234da167190a18b34efe6275622687500346ad3d42d9fad6b25d52d"
|
| 126 |
-
},
|
| 127 |
-
{
|
| 128 |
-
"file": "backbone/model-00003-of-00005.safetensors",
|
| 129 |
-
"bytes": 3937871784,
|
| 130 |
-
"sha256": "0d0af48a06e40b0afdcb7aa02cf6d8d6df885ccc9e2593e3db72cd3a92e798a7"
|
| 131 |
-
},
|
| 132 |
-
{
|
| 133 |
-
"file": "backbone/model-00004-of-00005.safetensors",
|
| 134 |
-
"bytes": 3926578128,
|
| 135 |
-
"sha256": "b960486d5d833e016c9e2a32d42214db14a8e48afef49d61a7f514f8c636a264"
|
| 136 |
-
},
|
| 137 |
-
{
|
| 138 |
-
"file": "backbone/model-00005-of-00005.safetensors",
|
| 139 |
-
"bytes": 976029360,
|
| 140 |
-
"sha256": "994dac35c408a89d89768febad1ed52a255da62f7924f86bd76371381e9cc751"
|
| 141 |
-
},
|
| 142 |
-
{
|
| 143 |
-
"file": "decision_config.json",
|
| 144 |
-
"bytes": 1070,
|
| 145 |
-
"sha256": "3324b9318a873fc27a8610e19d6d7ba82b0890a49011abbc8b097bdf429e6747"
|
| 146 |
-
},
|
| 147 |
-
{
|
| 148 |
-
"file": "decision_head.safetensors",
|
| 149 |
-
"bytes": 10529624,
|
| 150 |
-
"sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
|
| 151 |
-
},
|
| 152 |
-
{
|
| 153 |
-
"file": "tokenizer.json",
|
| 154 |
-
"bytes": 19989325,
|
| 155 |
-
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
|
| 156 |
-
},
|
| 157 |
-
{
|
| 158 |
-
"file": "tokenizer_config.json",
|
| 159 |
-
"bytes": 1123,
|
| 160 |
-
"sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
|
| 161 |
-
}
|
| 162 |
-
],
|
| 163 |
-
"source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
|
| 164 |
-
"source_api_code_sha256": "147b2fec32cbbbbb1b92cf2a19bb887f9d945e4974e141ca7ea93b8c542f5b21",
|
| 165 |
-
"dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
|
| 166 |
-
"production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
|
| 167 |
-
"temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
|
| 168 |
-
"runtime_sha256": "c5d3521358b2817f4e56ea8150c5c612139b4e2b2c5bb09f512d9a7c0b5298b9",
|
| 169 |
-
"production_batch_size": 8,
|
| 170 |
-
"input_length_limit": 16384,
|
| 171 |
-
"original_checkpoint_name": "winner",
|
| 172 |
-
"no_publication_performed": true
|
| 173 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
USAGE.md
DELETED
|
@@ -1,59 +0,0 @@
|
|
| 1 |
-
# Use Nox-4B
|
| 2 |
-
|
| 3 |
-
The examples below use the [official TypeSafe SDK](https://docs.typesafe.ai/sdk/python/usage) and the standard [SystemOne HTTP request](https://docs.typesafe.ai/api). Configure your endpoint to serve `Decision-1.0-Nox-4B`, then replace the example URL and API key. A Hugging Face model repository is a weights download, not an inference endpoint.
|
| 4 |
-
|
| 5 |
-
```bash
|
| 6 |
-
pip install typesafe-sdk
|
| 7 |
-
```
|
| 8 |
-
|
| 9 |
-
```python
|
| 10 |
-
from typesafe_sdk import Choice, Noul, TypeSafeClient
|
| 11 |
-
|
| 12 |
-
with TypeSafeClient(
|
| 13 |
-
api_key="YOUR_ENDPOINT_API_KEY",
|
| 14 |
-
base_url="https://your-decision-endpoint.example",
|
| 15 |
-
model="Decision-1.0-Nox-4B",
|
| 16 |
-
) as client:
|
| 17 |
-
result = client.system_one(
|
| 18 |
-
state="Customer reports a duplicate charge and asks for a refund.",
|
| 19 |
-
questions={
|
| 20 |
-
"route": Choice(
|
| 21 |
-
instructions="Which team should handle this request?",
|
| 22 |
-
criteria={"billing": "Payments and refunds", "technical": "Product faults"},
|
| 23 |
-
),
|
| 24 |
-
"refund_requested": Noul(instructions="Did the customer request a refund?"),
|
| 25 |
-
},
|
| 26 |
-
)
|
| 27 |
-
print(result.choices["route"].choice)
|
| 28 |
-
print(result.nouls["refund_requested"].noul)
|
| 29 |
-
```
|
| 30 |
-
|
| 31 |
-
```bash
|
| 32 |
-
curl -X POST 'https://your-decision-endpoint.example/v1/systemone' \
|
| 33 |
-
-H 'Authorization: Bearer YOUR_ENDPOINT_API_KEY' \
|
| 34 |
-
-H 'Content-Type: application/json' \
|
| 35 |
-
--data-raw '{
|
| 36 |
-
"model": "Decision-1.0-Nox-4B",
|
| 37 |
-
"state": "Customer reports a duplicate charge and asks for a refund.",
|
| 38 |
-
"questions": {
|
| 39 |
-
"route": {
|
| 40 |
-
"type": "choice",
|
| 41 |
-
"instructions": "Which team should handle this request?",
|
| 42 |
-
"criteria": {
|
| 43 |
-
"billing": "Payments and refunds",
|
| 44 |
-
"technical": "Product faults"
|
| 45 |
-
}
|
| 46 |
-
},
|
| 47 |
-
"refund_requested": {
|
| 48 |
-
"type": "noul",
|
| 49 |
-
"instructions": "Did the customer request a refund?"
|
| 50 |
-
}
|
| 51 |
-
}
|
| 52 |
-
}'
|
| 53 |
-
```
|
| 54 |
-
|
| 55 |
-
The state can be text or JSON-compatible structured data. Question IDs and Choice IDs are preserved in the response. Choice uses 2–255 options; Noul returns `noul`, the probability of a condition being true; Score uses 2–10 rubric descriptions ordered from index zero. A request can contain many questions.
|
| 56 |
-
|
| 57 |
-
Choice returns `choice`, `probabilities` and `confidence`. Score returns an expected zero-based `score`, a probability distribution and `legend`. Local Decision confidence is normalized maximum probability, `(K × max(p) − 1)/(K − 1)`; it does not reproduce an unpublished provider confidence statistic. An HTTP integration must supply `usage.input_tokens` and `usage.output_tokens`; counting generated tokens as zero is appropriate for this non-generative model, not a claim about provider billing.
|
| 58 |
-
|
| 59 |
-
The native runtime processes independent complete questions in batches of eight. Each complete rendered question, including state, instructions and criteria, must fit 16,384 tokens; overflow is rejected. [Runtime and hardware requirements](RUNTIME.md).
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
WEIGHTING.md
DELETED
|
@@ -1,21 +0,0 @@
|
|
| 1 |
-
# Weight sensitivity
|
| 2 |
-
|
| 3 |
-
The current product-priority weights were chosen after observing results. This comparison holds every model and prediction fixed; reweighting is not a training improvement.
|
| 4 |
-
|
| 5 |
-
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
-
|---|---:|---:|---:|
|
| 7 |
-
| Lux-9B | 77.40 | 77.07 | 79.69 |
|
| 8 |
-
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
-
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
-
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
| 11 |
-
| Qwen3.5-9B | 69.73 | 69.70 | 71.99 |
|
| 12 |
-
| Decider | 67.71 | 67.97 | 71.75 |
|
| 13 |
-
| Qwen3.5-4B | 67.29 | 67.24 | 70.25 |
|
| 14 |
-
| Sol-2B | 66.32 | 65.48 | 70.14 |
|
| 15 |
-
| Eos-0.8B | 61.89 | 61.19 | 65.99 |
|
| 16 |
-
| Kev-0.8B | 58.28 | 58.33 | 59.75 |
|
| 17 |
-
| Qwen3.5-2B | 57.24 | 57.20 | 60.54 |
|
| 18 |
-
| Kai-0.6B | 53.52 | 53.05 | 55.82 |
|
| 19 |
-
| Laya · English | 51.03 | 50.85 | 51.76 |
|
| 20 |
-
| Laya · Multilingual | 47.19 | 47.18 | 48.56 |
|
| 21 |
-
| Jev | 81.05 | 81.45 | 82.45 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
assets/decision-expanded-old_core-600px.png
DELETED
Git LFS Details
|
assets/decision-expanded-old_core.pdf
DELETED
|
Binary file (44.4 kB)
|
|
|
assets/decision-expanded-old_core.png
DELETED
Git LFS Details
|
assets/decision-expanded-old_core.svg
DELETED
assets/decision-expanded-overview-600px.png
DELETED
|
Binary file (94.9 kB)
|
|
|
assets/decision-expanded-overview.pdf
DELETED
|
Binary file (41.6 kB)
|
|
|
assets/decision-expanded-overview.png
DELETED
Git LFS Details
|
assets/decision-expanded-overview.svg
DELETED
assets/decision-expanded-ranking-600px.png
DELETED
|
Binary file (83.3 kB)
|
|
|
assets/decision-expanded-ranking.pdf
DELETED
|
Binary file (37.1 kB)
|
|
|
assets/decision-expanded-ranking.png
DELETED
Git LFS Details
|
assets/decision-expanded-ranking.svg
DELETED
assets/decision-expanded-v3_core-600px.png
DELETED
Git LFS Details
|
assets/decision-expanded-v3_core.pdf
DELETED
|
Binary file (44.5 kB)
|
|
|
assets/decision-expanded-v3_core.png
DELETED
Git LFS Details
|
assets/decision-expanded-v3_core.svg
DELETED
assets/decision-expanded-v4-600px.png
DELETED
|
Binary file (76.7 kB)
|
|
|
assets/decision-expanded-v4.pdf
DELETED
|
Binary file (37.7 kB)
|
|
|
assets/decision-expanded-v4.png
DELETED
Git LFS Details
|
assets/decision-expanded-v4.svg
DELETED
assets/decision-expanded-v5-600px.png
DELETED
|
Binary file (95.1 kB)
|
|
|
assets/decision-expanded-v5.pdf
DELETED
|
Binary file (40.2 kB)
|
|
|
assets/decision-expanded-v5.png
DELETED
Git LFS Details
|
assets/decision-expanded-v5.svg
DELETED
assets/decision-family-header.png
DELETED
Git LFS Details
|
assets/decision-matrix.pdf
DELETED
|
Binary file (29.4 kB)
|
|
|
assets/decision-matrix.svg
DELETED
assets/decision-question-scaling-600px.png
DELETED
|
Binary file (40.8 kB)
|
|
|
assets/decision-question-scaling.pdf
DELETED
|
Binary file (19 kB)
|
|
|
assets/decision-question-scaling.svg
DELETED
assets/decision-ranking.pdf
DELETED
|
Binary file (25.4 kB)
|
|
|
assets/decision-ranking.svg
DELETED
bundle-manifest.json
DELETED
|
@@ -1,243 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"format": "research-pointer-bundle-v1",
|
| 3 |
-
"status": "null-description-candidate-awaiting-public-proof-and-full-regression",
|
| 4 |
-
"files": [
|
| 5 |
-
{
|
| 6 |
-
"file": "NORMALIZATION_RUNTIME.md",
|
| 7 |
-
"bytes": 756,
|
| 8 |
-
"sha256": "cf8e6ce1f07687a68b6adeb98e6704b8f292cb6adc7b10e4b84e8e70aa5472e5"
|
| 9 |
-
},
|
| 10 |
-
{
|
| 11 |
-
"file": "NULL_DESCRIPTION_RENDERING.json",
|
| 12 |
-
"bytes": 514,
|
| 13 |
-
"sha256": "3b531cab60cba35648ab71fe4be7fd1af619f02ee337e3beed6296fdfc8b9ec8"
|
| 14 |
-
},
|
| 15 |
-
{
|
| 16 |
-
"file": "RUNTIME_BINDING.json",
|
| 17 |
-
"bytes": 2621,
|
| 18 |
-
"sha256": "fb76d9fc9147e15a678eb91dfc40d7039845a4eaebdecce041a5c9e76c3d3e0f"
|
| 19 |
-
},
|
| 20 |
-
{
|
| 21 |
-
"file": "SERVING_OPTIMIZATION.json",
|
| 22 |
-
"bytes": 1548,
|
| 23 |
-
"sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
|
| 24 |
-
},
|
| 25 |
-
{
|
| 26 |
-
"file": "SOURCE_BUNDLE_MANIFEST.json",
|
| 27 |
-
"bytes": 5637,
|
| 28 |
-
"sha256": "d437bc0149bcb8c9891fbb33c5abc4c2336c3a89ab6cfa0981da5a7f1d19f1f4"
|
| 29 |
-
},
|
| 30 |
-
{
|
| 31 |
-
"file": "backbone/config.json",
|
| 32 |
-
"bytes": 1978,
|
| 33 |
-
"sha256": "ae3a463b32e95b6cc207a7af4f1defb4195f388eb6f9ff19d2690b73d4966953"
|
| 34 |
-
},
|
| 35 |
-
{
|
| 36 |
-
"file": "backbone/model-00001-of-00003.safetensors",
|
| 37 |
-
"bytes": 3991295368,
|
| 38 |
-
"sha256": "5ebccc395ef9fa4d61c79894ecefcdaa2a319c1c154221e1ad73663b6f82aacc"
|
| 39 |
-
},
|
| 40 |
-
{
|
| 41 |
-
"file": "backbone/model-00002-of-00003.safetensors",
|
| 42 |
-
"bytes": 3979828128,
|
| 43 |
-
"sha256": "990e2e79fc1ab009df9846ef9fddb31f4b988589bbda84f4728dbaa4196d7fcd"
|
| 44 |
-
},
|
| 45 |
-
{
|
| 46 |
-
"file": "backbone/model-00003-of-00003.safetensors",
|
| 47 |
-
"bytes": 440425856,
|
| 48 |
-
"sha256": "c47859a3192bdae5003e4732fb911c8dddcf6e8caa167b4971e3a9801b828d64"
|
| 49 |
-
},
|
| 50 |
-
{
|
| 51 |
-
"file": "backbone/model.safetensors.index.json",
|
| 52 |
-
"bytes": 33047,
|
| 53 |
-
"sha256": "1602d52e38d81586af85bc4ce29ce082c5fc5877c763b1ebcab7545320016599"
|
| 54 |
-
},
|
| 55 |
-
{
|
| 56 |
-
"file": "chat_template.jinja",
|
| 57 |
-
"bytes": 7756,
|
| 58 |
-
"sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
|
| 59 |
-
},
|
| 60 |
-
{
|
| 61 |
-
"file": "code/decision_api.py",
|
| 62 |
-
"bytes": 10978,
|
| 63 |
-
"sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e"
|
| 64 |
-
},
|
| 65 |
-
{
|
| 66 |
-
"file": "code/decision_model.py",
|
| 67 |
-
"bytes": 10114,
|
| 68 |
-
"sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646"
|
| 69 |
-
},
|
| 70 |
-
{
|
| 71 |
-
"file": "code/predict.py",
|
| 72 |
-
"bytes": 3164,
|
| 73 |
-
"sha256": "02352e8385ab47157b6459910da54d962e5da4bb4940d571faedc86bc5da9aee"
|
| 74 |
-
},
|
| 75 |
-
{
|
| 76 |
-
"file": "code/profile_guard.py",
|
| 77 |
-
"bytes": 6184,
|
| 78 |
-
"sha256": "061d6ba3edf032038b074b061879a1ff81cfd2928fc0131c66b15563f7822509"
|
| 79 |
-
},
|
| 80 |
-
{
|
| 81 |
-
"file": "code/runtime_profile.py",
|
| 82 |
-
"bytes": 2857,
|
| 83 |
-
"sha256": "3e26ed65f2cd706def209761421cb8fc825281e5c1ed8f168350365ad803dc96"
|
| 84 |
-
},
|
| 85 |
-
{
|
| 86 |
-
"file": "decision_config.json",
|
| 87 |
-
"bytes": 753,
|
| 88 |
-
"sha256": "443a9b3f191a8387915c606de8303700fc1a069fe7c8ad46a0eba5528a2549a8"
|
| 89 |
-
},
|
| 90 |
-
{
|
| 91 |
-
"file": "decision_head.safetensors",
|
| 92 |
-
"bytes": 10529624,
|
| 93 |
-
"sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
|
| 94 |
-
},
|
| 95 |
-
{
|
| 96 |
-
"file": "pyproject.toml",
|
| 97 |
-
"bytes": 436,
|
| 98 |
-
"sha256": "135a9516e87ca2fb41ef54a974e29132256b6c2d528a1cbdea1001e28f306946"
|
| 99 |
-
},
|
| 100 |
-
{
|
| 101 |
-
"file": "runtime-profile/l2norm_fwd_kernel.json",
|
| 102 |
-
"bytes": 26330,
|
| 103 |
-
"sha256": "d7ed7c9962a48efdfa76ed695c8bdc0f56afe1397c202c936eb78315e87a26df"
|
| 104 |
-
},
|
| 105 |
-
{
|
| 106 |
-
"file": "runtime-profile/profile.json",
|
| 107 |
-
"bytes": 35040,
|
| 108 |
-
"sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850"
|
| 109 |
-
},
|
| 110 |
-
{
|
| 111 |
-
"file": "runtime.json",
|
| 112 |
-
"bytes": 1112,
|
| 113 |
-
"sha256": "78a2c21f8138beeffb3e6b1c11e03cb48972a78ca5ef3d1f58c50651031c34b6"
|
| 114 |
-
},
|
| 115 |
-
{
|
| 116 |
-
"file": "src/decision/__init__.py",
|
| 117 |
-
"bytes": 167,
|
| 118 |
-
"sha256": "70de37df98b6fc8e3b9f9d43935ba31a32350496214c11adf8c5fba72c433313"
|
| 119 |
-
},
|
| 120 |
-
{
|
| 121 |
-
"file": "src/decision/example.py",
|
| 122 |
-
"bytes": 3581,
|
| 123 |
-
"sha256": "a54dec885f92c2d38d07ef2333dff51965be52bc119f772cc74200ed314e69b6"
|
| 124 |
-
},
|
| 125 |
-
{
|
| 126 |
-
"file": "src/decision/model.py",
|
| 127 |
-
"bytes": 9415,
|
| 128 |
-
"sha256": "ba240d7493fc29203fe036966f0ab911200cd4a9252b50b05423409977639ee0"
|
| 129 |
-
},
|
| 130 |
-
{
|
| 131 |
-
"file": "temperature.json",
|
| 132 |
-
"bytes": 6083,
|
| 133 |
-
"sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d"
|
| 134 |
-
},
|
| 135 |
-
{
|
| 136 |
-
"file": "tokenizer.json",
|
| 137 |
-
"bytes": 19989325,
|
| 138 |
-
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
|
| 139 |
-
},
|
| 140 |
-
{
|
| 141 |
-
"file": "tokenizer_config.json",
|
| 142 |
-
"bytes": 1123,
|
| 143 |
-
"sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
|
| 144 |
-
}
|
| 145 |
-
],
|
| 146 |
-
"tensors": [
|
| 147 |
-
{
|
| 148 |
-
"file": "backbone/model-00001-of-00003.safetensors",
|
| 149 |
-
"elements": 1995638528,
|
| 150 |
-
"elements_by_dtype": {
|
| 151 |
-
"BF16": 1995638528
|
| 152 |
-
}
|
| 153 |
-
},
|
| 154 |
-
{
|
| 155 |
-
"file": "backbone/model-00002-of-00003.safetensors",
|
| 156 |
-
"elements": 1989900928,
|
| 157 |
-
"elements_by_dtype": {
|
| 158 |
-
"BF16": 1989900928
|
| 159 |
-
}
|
| 160 |
-
},
|
| 161 |
-
{
|
| 162 |
-
"file": "backbone/model-00003-of-00003.safetensors",
|
| 163 |
-
"elements": 220211840,
|
| 164 |
-
"elements_by_dtype": {
|
| 165 |
-
"BF16": 220211840
|
| 166 |
-
}
|
| 167 |
-
},
|
| 168 |
-
{
|
| 169 |
-
"file": "decision_head.safetensors",
|
| 170 |
-
"elements": 2632192,
|
| 171 |
-
"elements_by_dtype": {
|
| 172 |
-
"F32": 2632192
|
| 173 |
-
}
|
| 174 |
-
}
|
| 175 |
-
],
|
| 176 |
-
"source_checkpoint_files": [
|
| 177 |
-
{
|
| 178 |
-
"file": "backbone/config.json",
|
| 179 |
-
"bytes": 1977,
|
| 180 |
-
"sha256": "a5ed4156fda05f0f9149c66964d6165916754e7355488a8a07d9b0398acdbdb9"
|
| 181 |
-
},
|
| 182 |
-
{
|
| 183 |
-
"file": "backbone/model-00001-of-00005.safetensors",
|
| 184 |
-
"bytes": 3992269864,
|
| 185 |
-
"sha256": "d65565b8988a20b07988db09ed2321111e57b4fe656112fe4636715f8418c40b"
|
| 186 |
-
},
|
| 187 |
-
{
|
| 188 |
-
"file": "backbone/model-00002-of-00005.safetensors",
|
| 189 |
-
"bytes": 3990302320,
|
| 190 |
-
"sha256": "99197c2df234da167190a18b34efe6275622687500346ad3d42d9fad6b25d52d"
|
| 191 |
-
},
|
| 192 |
-
{
|
| 193 |
-
"file": "backbone/model-00003-of-00005.safetensors",
|
| 194 |
-
"bytes": 3937871784,
|
| 195 |
-
"sha256": "0d0af48a06e40b0afdcb7aa02cf6d8d6df885ccc9e2593e3db72cd3a92e798a7"
|
| 196 |
-
},
|
| 197 |
-
{
|
| 198 |
-
"file": "backbone/model-00004-of-00005.safetensors",
|
| 199 |
-
"bytes": 3926578128,
|
| 200 |
-
"sha256": "b960486d5d833e016c9e2a32d42214db14a8e48afef49d61a7f514f8c636a264"
|
| 201 |
-
},
|
| 202 |
-
{
|
| 203 |
-
"file": "backbone/model-00005-of-00005.safetensors",
|
| 204 |
-
"bytes": 976029360,
|
| 205 |
-
"sha256": "994dac35c408a89d89768febad1ed52a255da62f7924f86bd76371381e9cc751"
|
| 206 |
-
},
|
| 207 |
-
{
|
| 208 |
-
"file": "decision_config.json",
|
| 209 |
-
"bytes": 1070,
|
| 210 |
-
"sha256": "3324b9318a873fc27a8610e19d6d7ba82b0890a49011abbc8b097bdf429e6747"
|
| 211 |
-
},
|
| 212 |
-
{
|
| 213 |
-
"file": "decision_head.safetensors",
|
| 214 |
-
"bytes": 10529624,
|
| 215 |
-
"sha256": "9cb6f639714e31bcb76b58eaf94af0b72ac3db9d091ebcda45d9b575efd489de"
|
| 216 |
-
},
|
| 217 |
-
{
|
| 218 |
-
"file": "tokenizer.json",
|
| 219 |
-
"bytes": 19989325,
|
| 220 |
-
"sha256": "06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523"
|
| 221 |
-
},
|
| 222 |
-
{
|
| 223 |
-
"file": "tokenizer_config.json",
|
| 224 |
-
"bytes": 1123,
|
| 225 |
-
"sha256": "bee8eba30f0eb4af73c0fe2cd06d0f89b657d7819941c438157ec42f7c80ea87"
|
| 226 |
-
}
|
| 227 |
-
],
|
| 228 |
-
"source_model_code_sha256": "d3e28489c09f3bd7130e2d43d92e0b5c4a08e25b09b21303904defb0ff1c3646",
|
| 229 |
-
"source_api_code_sha256": "6716bdabca3cd2f1aa447d42a6cf53ee62d7ea97984a01f16b4c09fac8aaf17e",
|
| 230 |
-
"dev_sha256": "45b4cd46acc4be53b95c7ed8f333b3533972296a4d0557c7b8c48c9cab6ced61",
|
| 231 |
-
"production_predictions_sha256": "ba311f5a6bc73182651bdf5c9744f0ed95921110501ddc93dcfd20a1d06bceff",
|
| 232 |
-
"temperature_sha256": "69d80e5b215e2c1e6872f146fdb7ea5aa95c9fd678ef96208960d50e4e7b635d",
|
| 233 |
-
"runtime_sha256": "78a2c21f8138beeffb3e6b1c11e03cb48972a78ca5ef3d1f58c50651031c34b6",
|
| 234 |
-
"production_batch_size": 8,
|
| 235 |
-
"input_length_limit": 16384,
|
| 236 |
-
"original_checkpoint_name": "winner",
|
| 237 |
-
"no_publication_performed": true,
|
| 238 |
-
"source_bundle_manifest_sha256": "83876db506b2d98e3e8ce7d34310b21f97bac30d4f5aef3371053798bff08830",
|
| 239 |
-
"normalization_profile_sha256": "be32858d15233e0a3fbee0e4257fb02be0b3439deee4eb9c3f61151df7b73850",
|
| 240 |
-
"public_wrapper_included": true,
|
| 241 |
-
"source_published_bundle_manifest_sha256": "92d7f5be5e1ef21ee682f574b35cc01de4dd8ab16b8aa6edf0dfbdfcfcfabba3",
|
| 242 |
-
"serving_optimization_sha256": "fe2fd8c3e046faec139b8685eed693db0e6aa5235ebe5f860bc9f6dd1888c8ad"
|
| 243 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
code/decision_api.py
DELETED
|
@@ -1,198 +0,0 @@
|
|
| 1 |
-
"""Typed local inference adapter for the research decision checkpoints.
|
| 2 |
-
|
| 3 |
-
The response schema resembles TypeSafe's primitives. Confidence uses this
|
| 4 |
-
implementation's documented normalized maximum probability, not a claimed
|
| 5 |
-
reimplementation of TypeSafe's unpublished statistic. No text generation.
|
| 6 |
-
"""
|
| 7 |
-
from __future__ import annotations
|
| 8 |
-
import importlib.util
|
| 9 |
-
import math
|
| 10 |
-
from pathlib import Path
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
def prepare_runtime_profile(checkpoint, device='cuda:0'):
|
| 14 |
-
# This profile is verified before any dependency import can choose kernels.
|
| 15 |
-
import hashlib, json, sys
|
| 16 |
-
root=Path(checkpoint);runtime=json.loads((root/'runtime.json').read_text())
|
| 17 |
-
spec=runtime.get('normalization_profile')
|
| 18 |
-
if spec is None:
|
| 19 |
-
if '_decision_process_normalization_profile_v1' in sys.modules:
|
| 20 |
-
raise RuntimeError('Use separate processes for profiled and unprofiled models')
|
| 21 |
-
return None
|
| 22 |
-
import torch
|
| 23 |
-
target=torch.device(device)
|
| 24 |
-
if target.type!='cuda' or not torch.cuda.is_available():
|
| 25 |
-
raise RuntimeError('The bound profile requires a ROCm CUDA device')
|
| 26 |
-
arch=getattr(torch.cuda.get_device_properties(target),'gcnArchName','').split(':')[0]
|
| 27 |
-
if arch!=spec['validated_arch']:
|
| 28 |
-
raise RuntimeError('Target GPU architecture does not match the bound profile: '+arch)
|
| 29 |
-
relative=Path(spec['loader_file'])
|
| 30 |
-
if relative.is_absolute() or '..' in relative.parts:raise ValueError('Unsafe profile loader path')
|
| 31 |
-
path=root/relative
|
| 32 |
-
if hashlib.sha256(path.read_bytes()).hexdigest()!=spec['loader_sha256']:
|
| 33 |
-
raise ValueError('Bound runtime profile loader changed')
|
| 34 |
-
definition=importlib.util.spec_from_file_location('decision_bundle_runtime_profile',path)
|
| 35 |
-
module=importlib.util.module_from_spec(definition);definition.loader.exec_module(module)
|
| 36 |
-
return module.ensure_profile(root)
|
| 37 |
-
|
| 38 |
-
|
| 39 |
-
def question_row(state, name, question):
|
| 40 |
-
kind=question.get('type')
|
| 41 |
-
if kind not in {'choice','noul','score'}:raise ValueError('Unknown question type')
|
| 42 |
-
if 'instructions' not in question:raise ValueError('instructions is required')
|
| 43 |
-
criteria=question.get('criteria')
|
| 44 |
-
if kind=='noul':
|
| 45 |
-
criteria={} if criteria is None else criteria
|
| 46 |
-
if not isinstance(criteria,dict) or set(criteria)-{'true','false'}:
|
| 47 |
-
raise ValueError('noul criteria may contain only true and false')
|
| 48 |
-
options=[{'key':'false','description':criteria.get('false','The answer to the question is no.')},
|
| 49 |
-
{'key':'true','description':criteria.get('true','The answer to the question is yes.')}]
|
| 50 |
-
elif kind=='score':
|
| 51 |
-
if not isinstance(criteria,list) or not 2<=len(criteria)<=10:
|
| 52 |
-
raise ValueError('score requires an ordered list of 2..10 criteria')
|
| 53 |
-
options=[{'key':str(i),'description':value} for i,value in enumerate(criteria)]
|
| 54 |
-
else:
|
| 55 |
-
if not isinstance(criteria,dict) or not 2<=len(criteria)<=255:
|
| 56 |
-
raise ValueError('choice requires a mapping of 2..255 criteria')
|
| 57 |
-
if not all(isinstance(k,str) for k in criteria):raise ValueError('Choice keys must be strings')
|
| 58 |
-
options=[{'key':key,'description':key if value is None else value} for key,value in criteria.items()]
|
| 59 |
-
# The question name is used for bookkeeping only; encoders never render id.
|
| 60 |
-
return {'id':name,'state':state,'instructions':question['instructions'],
|
| 61 |
-
'options':options,'task_type':kind,'family':'inference'}
|
| 62 |
-
|
| 63 |
-
|
| 64 |
-
def typed_answer(row, probabilities):
|
| 65 |
-
p=[float(v) for v in probabilities];k=len(row['options'])
|
| 66 |
-
if len(p)!=k or any(not math.isfinite(v) or v<0 for v in p):
|
| 67 |
-
raise ValueError('Invalid probability vector')
|
| 68 |
-
total=sum(p)
|
| 69 |
-
if total<=0 or abs(total-1)>1e-4:raise ValueError('Probabilities must sum to one')
|
| 70 |
-
p=[v/total for v in p];selected=max(range(k),key=p.__getitem__)
|
| 71 |
-
kind=row['task_type']
|
| 72 |
-
if kind=='noul':
|
| 73 |
-
keys=[o['key'] for o in row['options']]
|
| 74 |
-
if set(keys)!={'false','true'}:raise ValueError('Native noul rows require false/true keys')
|
| 75 |
-
return {'type':'noul','noul':p[keys.index('true')]}
|
| 76 |
-
answer={'type':kind,'probabilities':{o['key']:v for o,v in zip(row['options'],p)},
|
| 77 |
-
'confidence':max(0.,min(1.,(k*max(p)-1)/(k-1)))}
|
| 78 |
-
if kind=='choice':answer['choice']=row['options'][selected]['key']
|
| 79 |
-
else:
|
| 80 |
-
if [o['key'] for o in row['options']] != [str(i) for i in range(k)]:
|
| 81 |
-
raise ValueError('Native score rows require ordered numeric level keys')
|
| 82 |
-
answer['score']=sum(i*v for i,v in enumerate(p))
|
| 83 |
-
answer['legend']={str(i):o['description'] for i,o in enumerate(row['options'])}
|
| 84 |
-
return answer
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
class DecisionEngine:
|
| 88 |
-
def __init__(self, checkpoint, model_code, *, device='cuda:0', max_length=16384,
|
| 89 |
-
batch_size=8, temperatures=None, model_name='local-decision-research'):
|
| 90 |
-
self.normalization_profile=prepare_runtime_profile(checkpoint, device=device)
|
| 91 |
-
import torch
|
| 92 |
-
path=Path(model_code)/'decision_model.py'
|
| 93 |
-
spec=importlib.util.spec_from_file_location('research_decision_runtime',path)
|
| 94 |
-
module=importlib.util.module_from_spec(spec);spec.loader.exec_module(module)
|
| 95 |
-
model,tokenizer=module.DecisionModel.from_checkpoint(checkpoint,dtype=torch.bfloat16)
|
| 96 |
-
self.model=model.to(device).eval();self.tokenizer=tokenizer;self.module=module
|
| 97 |
-
self.device=device;self.max_length=max_length;self.batch_size=batch_size
|
| 98 |
-
self.temperatures=temperatures or {};self.model_name=model_name
|
| 99 |
-
if batch_size<1 or max_length<1:raise ValueError('Positive batch_size/max_length required')
|
| 100 |
-
if any(not math.isfinite(v) or v<=0 for v in self.temperatures.values()):
|
| 101 |
-
raise ValueError('Temperatures must be finite positive numbers')
|
| 102 |
-
|
| 103 |
-
def predict_rows(self, rows):
|
| 104 |
-
import torch
|
| 105 |
-
encoded=encode_request(rows,self.tokenizer,self.module,self.max_length)
|
| 106 |
-
pad=self.tokenizer.pad_token_id if self.tokenizer.pad_token_id is not None else self.tokenizer.eos_token_id
|
| 107 |
-
records=[]
|
| 108 |
-
with torch.inference_mode():
|
| 109 |
-
for start in range(0,len(rows),self.batch_size):
|
| 110 |
-
items=encoded[start:start+self.batch_size]
|
| 111 |
-
batch={key:value.to(self.device) if torch.is_tensor(value) else value
|
| 112 |
-
for key,value in self.module.collate(items,pad).items()}
|
| 113 |
-
with torch.autocast('cuda',dtype=torch.bfloat16):logits=self.model(**batch)
|
| 114 |
-
# Preserve each row's original float/temperature/softmax math,
|
| 115 |
-
# but defer host synchronization until the complete batch.
|
| 116 |
-
staged=[];transfers=[]
|
| 117 |
-
for row,item,values in zip(rows[start:start+self.batch_size],items,logits):
|
| 118 |
-
k=len(row['options']);values=values[:k].float()
|
| 119 |
-
temperature=self.temperatures.get(row['task_type'],1.)
|
| 120 |
-
probabilities=(values/temperature).softmax(-1)
|
| 121 |
-
staged.append((row,item,k,temperature))
|
| 122 |
-
transfers.extend((values,probabilities))
|
| 123 |
-
host_values=torch.cat(transfers).tolist()
|
| 124 |
-
offset=0
|
| 125 |
-
for row,item,k,temperature in staged:
|
| 126 |
-
values=host_values[offset:offset+k]
|
| 127 |
-
probabilities=host_values[offset+k:offset+2*k]
|
| 128 |
-
offset+=2*k
|
| 129 |
-
answer=typed_answer(row,probabilities)
|
| 130 |
-
prediction=max(range(k),key=probabilities.__getitem__)
|
| 131 |
-
if row['task_type']=='noul':
|
| 132 |
-
chosen='true' if answer['noul']>=.5 else 'false'
|
| 133 |
-
prediction=[o['key'] for o in row['options']].index(chosen)
|
| 134 |
-
rec={'id':row['id'],'status':'ok','prediction':prediction,
|
| 135 |
-
'probabilities':probabilities,'logits':values,'temperature':temperature,
|
| 136 |
-
'native_contract':True,'truncated':False,'input_tokens':len(item['ids']),
|
| 137 |
-
'prompt_sha256':item['prompt_sha256'],'answer':answer}
|
| 138 |
-
if row['task_type']=='noul':rec['native_noul']=answer['noul']
|
| 139 |
-
if row['task_type']=='score':rec['native_score']=answer['score']
|
| 140 |
-
records.append(rec)
|
| 141 |
-
return records
|
| 142 |
-
|
| 143 |
-
def decide(self, state, questions):
|
| 144 |
-
if not isinstance(questions,dict) or not questions:
|
| 145 |
-
raise ValueError('questions must be a nonempty mapping')
|
| 146 |
-
if not all(isinstance(name,str) for name in questions):raise ValueError('Question names must be strings')
|
| 147 |
-
rows=[question_row(state,name,q) for name,q in questions.items()]
|
| 148 |
-
result=self.predict_rows(rows)
|
| 149 |
-
return {'model':self.model_name,'answers':{r['id']:r['answer'] for r in result},
|
| 150 |
-
'usage':{'input_tokens':sum(r['input_tokens'] for r in result),'scored_questions':len(result)}}
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
"""Experimental request-local exact-segment tokenization.
|
| 154 |
-
|
| 155 |
-
Original encode/segments functions remain authoritative. Batch tokenize exact
|
| 156 |
-
whole segments, never split a BPE prefix at a new boundary. No cross-request
|
| 157 |
-
cache, GPU change, prompt change or change to the eight-row inference groups.
|
| 158 |
-
"""
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
class SegmentLookup:
|
| 162 |
-
def __init__(self, tokenizer, cache):
|
| 163 |
-
self.tokenizer, self.cache = tokenizer, cache
|
| 164 |
-
|
| 165 |
-
def encode(self, text, **kwargs):
|
| 166 |
-
if kwargs == {'add_special_tokens': False} and text in self.cache:
|
| 167 |
-
# Original encode extends its prefix list in place.
|
| 168 |
-
return list(self.cache[text])
|
| 169 |
-
return self.tokenizer.encode(text, **kwargs)
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
def encode_request(rows, tokenizer, module, max_length=16384,
|
| 173 |
-
max_cached_characters=8_000_000, segment_batch_size=64):
|
| 174 |
-
if max_cached_characters < 0 or segment_batch_size < 1:
|
| 175 |
-
raise ValueError('Invalid tokenizer resource bound')
|
| 176 |
-
unique = {}
|
| 177 |
-
characters = 0
|
| 178 |
-
for row in rows:
|
| 179 |
-
prefix, options, suffix = module.segments(row)
|
| 180 |
-
for segment in (prefix, *options, suffix):
|
| 181 |
-
if segment not in unique:
|
| 182 |
-
unique[segment] = None
|
| 183 |
-
characters += len(segment)
|
| 184 |
-
if characters > max_cached_characters:
|
| 185 |
-
# Preserve the original behavior under the resource cap.
|
| 186 |
-
return [module.encode(r, tokenizer, max_length) for r in rows]
|
| 187 |
-
strings = list(unique)
|
| 188 |
-
for start in range(0, len(strings), segment_batch_size):
|
| 189 |
-
batch = strings[start:start + segment_batch_size]
|
| 190 |
-
result = tokenizer(batch, add_special_tokens=False, padding=False,
|
| 191 |
-
truncation=False, return_attention_mask=False,
|
| 192 |
-
return_token_type_ids=False)['input_ids']
|
| 193 |
-
if len(result) != len(batch):
|
| 194 |
-
raise ValueError('Batch tokenizer output count differs')
|
| 195 |
-
for segment, ids in zip(batch, result):
|
| 196 |
-
unique[segment] = tuple(ids)
|
| 197 |
-
lookup = SegmentLookup(tokenizer, unique)
|
| 198 |
-
return [module.encode(row, lookup, max_length) for row in rows]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
code/decision_model.py
DELETED
|
@@ -1,173 +0,0 @@
|
|
| 1 |
-
"""Dynamic candidate readout over a causal Qwen3.5 text backbone.
|
| 2 |
-
|
| 3 |
-
Candidate endpoints retain their contextual vectors. A final global-query
|
| 4 |
-
vector can incorporate all options before a shared bilinear + MLP scorer
|
| 5 |
-
scores every candidate. This is a research architecture, not a Jev claim.
|
| 6 |
-
"""
|
| 7 |
-
import hashlib
|
| 8 |
-
import json
|
| 9 |
-
import math
|
| 10 |
-
from pathlib import Path
|
| 11 |
-
|
| 12 |
-
import torch
|
| 13 |
-
from torch import nn
|
| 14 |
-
import torch.nn.functional as F
|
| 15 |
-
from safetensors.torch import load_file, save_file
|
| 16 |
-
from transformers import AutoTokenizer, Qwen3_5ForConditionalGeneration
|
| 17 |
-
from transformers.models.qwen3_5.modeling_qwen3_5 import Qwen3_5TextModel
|
| 18 |
-
|
| 19 |
-
PROMPT_VERSION = "structured-segmented-candidate-endpoints-global-query-v2"
|
| 20 |
-
MAX_OPTIONS = 255
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
def canonical(value):
|
| 24 |
-
return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
def payload(value):
|
| 28 |
-
return value if isinstance(value, str) else canonical(value)
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
def segments(row):
|
| 32 |
-
opts = row["options"]
|
| 33 |
-
if not 2 <= len(opts) <= MAX_OPTIONS:
|
| 34 |
-
raise ValueError(f"{row['id']}: expected 2..255 options")
|
| 35 |
-
if not all(isinstance(o["key"], str) for o in opts):
|
| 36 |
-
raise ValueError("Option keys must be strings")
|
| 37 |
-
if len({o["key"] for o in opts}) != len(opts):
|
| 38 |
-
raise ValueError("Duplicate option keys")
|
| 39 |
-
prefix = f"Context:\n{payload(row['state'])}\n\nTask type: {row.get('task_type', 'choice')}\nQuestion:\n{payload(row['instructions'])}\nOptions:"
|
| 40 |
-
# Tokenize each part separately. This deliberately fixes boundaries and
|
| 41 |
-
# avoids guessing endpoint indices from merged BPE character offsets.
|
| 42 |
-
options = ["\n<option>\n" + canonical({"key": o["key"], "description": o.get("description")}) + "\n</option>" for o in opts]
|
| 43 |
-
suffix = "\n\nSelect the single option best supported by the context and instructions.\nDecision:"
|
| 44 |
-
return prefix, options, suffix
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
def render(row):
|
| 48 |
-
prefix, opts, suffix = segments(row)
|
| 49 |
-
return prefix + "".join(opts) + suffix
|
| 50 |
-
|
| 51 |
-
|
| 52 |
-
def encode(row, tokenizer, max_length=16384):
|
| 53 |
-
prefix, opts, suffix = segments(row)
|
| 54 |
-
ids = tokenizer.encode(prefix, add_special_tokens=False)
|
| 55 |
-
candidate_positions = []
|
| 56 |
-
for option in opts:
|
| 57 |
-
part = tokenizer.encode(option, add_special_tokens=False)
|
| 58 |
-
if not part:
|
| 59 |
-
raise ValueError("Empty tokenized candidate")
|
| 60 |
-
ids.extend(part)
|
| 61 |
-
candidate_positions.append(len(ids) - 1)
|
| 62 |
-
ids.extend(tokenizer.encode(suffix, add_special_tokens=False))
|
| 63 |
-
if len(ids) > max_length:
|
| 64 |
-
raise ValueError(f"{row['id']}: {len(ids)} tokens exceeds max_length={max_length}; no truncation allowed")
|
| 65 |
-
label = row.get("label", -1)
|
| 66 |
-
if label != -1 and not 0 <= label < len(opts):
|
| 67 |
-
raise ValueError("Invalid label")
|
| 68 |
-
prompt = prefix + "".join(opts) + suffix
|
| 69 |
-
return {"id": row["id"], "ids": ids, "label": label, "nopts": len(opts), "family": row.get("family", "unspecified"), "candidate_positions": candidate_positions, "query_position": len(ids) - 1, "target_probs": row.get("target_probs"), "prompt_sha256": hashlib.sha256(prompt.encode()).hexdigest(), "token_ids_sha256": hashlib.sha256(canonical(ids).encode()).hexdigest(), "segmented_tokenization": True}
|
| 70 |
-
|
| 71 |
-
|
| 72 |
-
def collate(items, pad_id):
|
| 73 |
-
length = ((max(len(x["ids"]) for x in items) + 31) // 32) * 32
|
| 74 |
-
nopts = max(x["nopts"] for x in items)
|
| 75 |
-
ids = torch.full((len(items), length), pad_id, dtype=torch.long)
|
| 76 |
-
mask = torch.zeros_like(ids)
|
| 77 |
-
positions = torch.zeros((len(items), nopts), dtype=torch.long)
|
| 78 |
-
candidate_mask = torch.zeros((len(items), nopts), dtype=torch.bool)
|
| 79 |
-
for i, item in enumerate(items):
|
| 80 |
-
if len(item["candidate_positions"]) != item["nopts"]:
|
| 81 |
-
raise ValueError("Candidate count does not match endpoint count")
|
| 82 |
-
if not all(0 <= p < item["query_position"] < len(item["ids"]) for p in item["candidate_positions"]):
|
| 83 |
-
raise ValueError("Candidate endpoints must precede global query")
|
| 84 |
-
if len(set(item["candidate_positions"])) != item["nopts"]:
|
| 85 |
-
raise ValueError("Duplicate candidate endpoint")
|
| 86 |
-
ids[i, :len(item["ids"])] = torch.tensor(item["ids"])
|
| 87 |
-
mask[i, :len(item["ids"])] = 1
|
| 88 |
-
positions[i, :item["nopts"]] = torch.tensor(item["candidate_positions"])
|
| 89 |
-
candidate_mask[i, :item["nopts"]] = True
|
| 90 |
-
return {"input_ids": ids, "attention_mask": mask, "candidate_positions": positions, "candidate_mask": candidate_mask, "query_positions": torch.tensor([x["query_position"] for x in items]), "labels": torch.tensor([x["label"] for x in items]), "nopts": torch.tensor([x["nopts"] for x in items]), "ids": [x["id"] for x in items], "families": [x["family"] for x in items]}
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
class CandidateHead(nn.Module):
|
| 94 |
-
def __init__(self, hidden_size, head_dim=256):
|
| 95 |
-
super().__init__()
|
| 96 |
-
self.head_dim = head_dim
|
| 97 |
-
self.candidate_norm = nn.LayerNorm(hidden_size)
|
| 98 |
-
self.query_norm = nn.LayerNorm(hidden_size)
|
| 99 |
-
self.key = nn.Linear(hidden_size, head_dim, bias=False)
|
| 100 |
-
self.query = nn.Linear(hidden_size, head_dim, bias=False)
|
| 101 |
-
self.candidate_mlp = nn.Linear(hidden_size, head_dim, bias=True)
|
| 102 |
-
self.query_mlp = nn.Linear(hidden_size, head_dim, bias=False)
|
| 103 |
-
self.scalar = nn.Linear(head_dim, 1, bias=False)
|
| 104 |
-
nn.init.normal_(self.scalar.weight, mean=0., std=0.01)
|
| 105 |
-
|
| 106 |
-
def forward(self, candidates, query):
|
| 107 |
-
# Keep the small shared head in FP32 even when the backbone uses BF16.
|
| 108 |
-
# The v1 letter head exhibited BF16 ties sensitive to batch padding.
|
| 109 |
-
with torch.autocast(device_type=candidates.device.type, enabled=False):
|
| 110 |
-
c = self.candidate_norm(candidates.float())
|
| 111 |
-
q = self.query_norm(query.float())
|
| 112 |
-
bilinear = (self.key(c) * self.query(q)[:, None, :]).sum(-1) / math.sqrt(self.head_dim)
|
| 113 |
-
interaction = self.scalar(F.gelu(self.candidate_mlp(c) + self.query_mlp(q)[:, None, :])).squeeze(-1)
|
| 114 |
-
return bilinear + interaction
|
| 115 |
-
|
| 116 |
-
|
| 117 |
-
class DecisionModel(nn.Module):
|
| 118 |
-
def __init__(self, backbone, head, metadata):
|
| 119 |
-
super().__init__()
|
| 120 |
-
self.backbone, self.head, self.metadata = backbone, head, metadata
|
| 121 |
-
|
| 122 |
-
@classmethod
|
| 123 |
-
def from_base(cls, path, revision="local", dtype=torch.bfloat16, attention="sdpa", head_dim=256):
|
| 124 |
-
tokenizer = AutoTokenizer.from_pretrained(path, local_files_only=True)
|
| 125 |
-
full, info = Qwen3_5ForConditionalGeneration.from_pretrained(path, dtype=dtype, local_files_only=True, attn_implementation=attention, output_loading_info=True)
|
| 126 |
-
if any(info.get(k) for k in ("missing_keys", "mismatched_keys", "error_msgs")):
|
| 127 |
-
raise RuntimeError(f"Incomplete base loading: {info}")
|
| 128 |
-
backbone = full.model.language_model
|
| 129 |
-
backbone.config.use_cache = False
|
| 130 |
-
head = CandidateHead(backbone.config.hidden_size, head_dim)
|
| 131 |
-
metadata = {"base_revision": revision, "text_parameter_count": sum(p.numel() for p in backbone.parameters()), "prompt_version": PROMPT_VERSION, "attention": attention, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "head_precision": "float32-outside-autocast"}
|
| 132 |
-
return cls(backbone, head, metadata), tokenizer
|
| 133 |
-
|
| 134 |
-
@classmethod
|
| 135 |
-
def from_decision_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa", head_dim=256):
|
| 136 |
-
"""Warm-start the backbone of a trained v1 model; initialize a new head."""
|
| 137 |
-
path = Path(path)
|
| 138 |
-
metadata = json.loads((path / "decision_config.json").read_text())
|
| 139 |
-
backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention)
|
| 140 |
-
backbone.config.use_cache = False
|
| 141 |
-
head = CandidateHead(backbone.config.hidden_size, head_dim)
|
| 142 |
-
metadata.update({"prompt_version": PROMPT_VERSION, "head_dim": head_dim, "max_options": MAX_OPTIONS, "architecture": "contextual-candidate-endpoint-plus-global-query-shared-bilinear-mlp", "head_initialization": "random-shared-content-scorer", "warm_start": "trained-v1-text-backbone", "head_precision": "float32-outside-autocast"})
|
| 143 |
-
return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True)
|
| 144 |
-
|
| 145 |
-
@classmethod
|
| 146 |
-
def from_checkpoint(cls, path, dtype=torch.bfloat16, attention="sdpa"):
|
| 147 |
-
path = Path(path)
|
| 148 |
-
metadata = json.loads((path / "decision_config.json").read_text())
|
| 149 |
-
if metadata["prompt_version"] != PROMPT_VERSION:
|
| 150 |
-
raise ValueError("Not a pointer-v2 checkpoint; use from_decision_checkpoint for warm start")
|
| 151 |
-
backbone = Qwen3_5TextModel.from_pretrained(path / "backbone", dtype=dtype, local_files_only=True, attn_implementation=attention)
|
| 152 |
-
head = CandidateHead(backbone.config.hidden_size, metadata["head_dim"])
|
| 153 |
-
head.load_state_dict(load_file(path / "decision_head.safetensors"))
|
| 154 |
-
return cls(backbone, head, metadata), AutoTokenizer.from_pretrained(path, local_files_only=True)
|
| 155 |
-
|
| 156 |
-
def forward(self, input_ids, attention_mask, candidate_positions, candidate_mask, query_positions, **unused):
|
| 157 |
-
hidden = self.backbone(input_ids=input_ids, attention_mask=attention_mask, use_cache=False).last_hidden_state
|
| 158 |
-
batches = torch.arange(hidden.shape[0], device=hidden.device)
|
| 159 |
-
candidates = hidden[batches[:, None], candidate_positions]
|
| 160 |
-
query = hidden[batches, query_positions]
|
| 161 |
-
scores = self.head(candidates, query).float()
|
| 162 |
-
return scores.masked_fill(~candidate_mask, -float("inf"))
|
| 163 |
-
|
| 164 |
-
def save(self, path, tokenizer):
|
| 165 |
-
path = Path(path); path.mkdir(parents=True, exist_ok=True)
|
| 166 |
-
self.backbone.save_pretrained(path / "backbone", safe_serialization=True, max_shard_size="4GB")
|
| 167 |
-
save_file({n: v.detach().cpu().contiguous() for n, v in self.head.state_dict().items()}, str(path / "decision_head.safetensors"))
|
| 168 |
-
tokenizer.save_pretrained(path)
|
| 169 |
-
(path / "decision_config.json").write_text(json.dumps(self.metadata, indent=2) + "\n")
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
def classification_loss(logits, labels):
|
| 173 |
-
return F.cross_entropy(logits, labels)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
code/predict.py
DELETED
|
@@ -1,43 +0,0 @@
|
|
| 1 |
-
"""Portable one-GPU JSONL inference for frozen dynamic-option checkpoints."""
|
| 2 |
-
import argparse
|
| 3 |
-
import hashlib
|
| 4 |
-
import json
|
| 5 |
-
from pathlib import Path
|
| 6 |
-
import time
|
| 7 |
-
import torch
|
| 8 |
-
from decision_model import DecisionModel, encode, collate
|
| 9 |
-
|
| 10 |
-
p = argparse.ArgumentParser()
|
| 11 |
-
p.add_argument("--model", required=True)
|
| 12 |
-
p.add_argument("--base", action="store_true")
|
| 13 |
-
p.add_argument("--revision", default="local-checkpoint")
|
| 14 |
-
p.add_argument("--input", required=True)
|
| 15 |
-
p.add_argument("--output", required=True)
|
| 16 |
-
p.add_argument("--batch-size", type=int, default=8)
|
| 17 |
-
p.add_argument("--max-length", type=int, default=4096)
|
| 18 |
-
p.add_argument("--temperature", type=float, default=1.0)
|
| 19 |
-
a = p.parse_args()
|
| 20 |
-
assert a.temperature > 0
|
| 21 |
-
torch.cuda.set_device(0)
|
| 22 |
-
model, tokenizer = DecisionModel.from_base(a.model, a.revision) if a.base else DecisionModel.from_checkpoint(a.model)
|
| 23 |
-
model = model.cuda().eval()
|
| 24 |
-
rows = [json.loads(line) for line in Path(a.input).read_text().splitlines() if line.strip()]
|
| 25 |
-
encoded = [encode(row, tokenizer, a.max_length) for row in rows]
|
| 26 |
-
assert len({row["id"] for row in rows}) == len(rows)
|
| 27 |
-
output = Path(a.output); output.parent.mkdir(parents=True, exist_ok=True)
|
| 28 |
-
if output.exists(): raise RuntimeError("Refusing to overwrite predictions")
|
| 29 |
-
with torch.inference_mode(), output.open("w") as f:
|
| 30 |
-
for start in range(0, len(encoded), a.batch_size):
|
| 31 |
-
examples = encoded[start:start + a.batch_size]
|
| 32 |
-
batch = {key: value.cuda() if torch.is_tensor(value) else value for key, value in collate(examples, tokenizer.pad_token_id if tokenizer.pad_token_id is not None else tokenizer.eos_token_id).items()}
|
| 33 |
-
torch.cuda.synchronize(); tick = time.perf_counter()
|
| 34 |
-
with torch.autocast("cuda", dtype=torch.bfloat16): logits = model(**batch)
|
| 35 |
-
torch.cuda.synchronize(); elapsed = time.perf_counter() - tick
|
| 36 |
-
probabilities = (logits / a.temperature).softmax(-1).cpu().tolist(); scores = logits.cpu().tolist()
|
| 37 |
-
for row, example, prob, score in zip(rows[start:start + a.batch_size], examples, probabilities, scores):
|
| 38 |
-
k = example["nopts"]; pred = max(range(k), key=lambda i: prob[i])
|
| 39 |
-
rec = {"id": row["id"], "family": row.get("family"), "label": row.get("label"), "prediction": pred, "prediction_key": row["options"][pred]["key"], "probabilities": prob[:k], "logits": score[:k], "temperature": a.temperature, "input_tokens": len(example["ids"]), "prompt_sha256": example["prompt_sha256"], "batch_elapsed_seconds": elapsed, "batch_size": len(examples)}
|
| 40 |
-
if "score_values" in row: rec["expected_score"] = sum(value * probability for value, probability in zip(row["score_values"], prob[:k]))
|
| 41 |
-
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
| 42 |
-
f.flush(); print(json.dumps({"completed": min(start + a.batch_size, len(rows)), "total": len(rows)}), flush=True)
|
| 43 |
-
Path(str(output) + ".metadata.json").write_text(json.dumps({"model": a.model, "revision": a.revision, "temperature": a.temperature, "input_sha256": hashlib.sha256(Path(a.input).read_bytes()).hexdigest(), "predictions_sha256": hashlib.sha256(output.read_bytes()).hexdigest(), "rows": len(rows), "model_metadata": model.metadata}, indent=2))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
code/profile_guard.py
DELETED
|
@@ -1,82 +0,0 @@
|
|
| 1 |
-
"""Fail-closed official FLA strict-config setup for one isolated process.
|
| 2 |
-
|
| 3 |
-
No model/prompt/head changes. The guard prevents FLA STRICT's ordinary missing
|
| 4 |
-
key fallback and records actual configured calls. This is a diagnostic module,
|
| 5 |
-
not an installed change to the published wrapper or dependency environment.
|
| 6 |
-
"""
|
| 7 |
-
import hashlib,importlib,json,os,sys
|
| 8 |
-
from pathlib import Path
|
| 9 |
-
|
| 10 |
-
def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
|
| 11 |
-
def serialized(key):return json.dumps(key,separators=(',',':'),sort_keys=True)
|
| 12 |
-
def config_fields(config):
|
| 13 |
-
if isinstance(config,dict):return {k:config.get(k) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
|
| 14 |
-
return {k:getattr(config,k,None) for k in ['kwargs','num_warps','num_stages','num_ctas','maxnreg','ir_override']}
|
| 15 |
-
def validate_profile(path,expected_sha):
|
| 16 |
-
path=Path(path).resolve()
|
| 17 |
-
if sha(path)!=expected_sha:raise ValueError('Profile hash changed')
|
| 18 |
-
profile=json.loads(path.read_text())
|
| 19 |
-
if profile['format']!='decision-fla-l2norm-profile-v1' or profile.get('model_family')!='Qwen/Qwen3.5-4B' or profile['cache_mode']!='strict':raise ValueError('Profile format/mode unsupported')
|
| 20 |
-
if len(profile['files'])!=1 or profile['files'][0]['file']!='l2norm_fwd_kernel.json':raise ValueError('Unexpected profile file set')
|
| 21 |
-
f=path.parent/'l2norm_fwd_kernel.json'
|
| 22 |
-
if sha(f)!=profile['files'][0]['sha256']:raise ValueError('Explicit kernel config changed')
|
| 23 |
-
data=json.loads(f.read_text());entries={}
|
| 24 |
-
if data.get('default_config') is not None:raise ValueError('Implicit fallback defaults forbidden')
|
| 25 |
-
for h,item in data['autotune_entries'].items():
|
| 26 |
-
key=item['autotune_key'];encoded=serialized(key)
|
| 27 |
-
if hashlib.md5(encoded.encode()).hexdigest()!=h or encoded in entries:raise ValueError('Invalid/duplicate key')
|
| 28 |
-
if len(key)!=5 or key[0]!=128 or type(key[1]) is not int or not 1<=key[1]<=64 or key[2:]!=['torch.bfloat16','torch.bfloat16','torch.float32']:raise ValueError('Unsupported numerical key')
|
| 29 |
-
c=item['config']
|
| 30 |
-
if c['kwargs'].keys()!={'BT'} or c['kwargs']['BT'] not in [8,16,32,64] or c['num_warps'] not in [1,2,4,8,16] or c['num_stages']!=3 or c['num_ctas']!=1 or any(c.get(x) is not None for x in ['maxnreg','pre_hook','ir_override']):raise ValueError('Unexpected launch configuration')
|
| 31 |
-
entries[encoded]=c
|
| 32 |
-
if {json.loads(k)[1] for k in entries}!=set(range(1,65)):raise ValueError('Incomplete legal NB coverage')
|
| 33 |
-
return path,profile,entries
|
| 34 |
-
|
| 35 |
-
def attach_guard(kernel,cache_module,entries,telemetry):
|
| 36 |
-
original=kernel.run
|
| 37 |
-
def guarded(*args,**kwargs):
|
| 38 |
-
if cache_module.FLA_CACHE_MODE is not cache_module.FlaCacheMode.STRICT:raise RuntimeError('FLA strict mode changed')
|
| 39 |
-
key=cache_module.AutotuneKey.build(kernel.arg_names,kernel.keys,args,kwargs);encoded=serialized(list(key.autotune_key))
|
| 40 |
-
if encoded not in entries:raise RuntimeError('Uncontracted FLA l2norm key: '+encoded)
|
| 41 |
-
expected=entries[encoded];loaded=cache_module.load_cached_config(kernel.kernel_name,key)
|
| 42 |
-
if config_fields(loaded)!=config_fields(expected):raise RuntimeError('FLA exact config lookup mismatch')
|
| 43 |
-
if key.autotune_key in kernel.cache and config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('A conflicting in-process kernel cache exists')
|
| 44 |
-
# Explicit official configuration load guarantees that the following original
|
| 45 |
-
# run finds this exact cache entry and cannot perform timing-based autotune.
|
| 46 |
-
kernel.maybe_load_cached_config(key)
|
| 47 |
-
if key.autotune_key not in kernel.cache or config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Official strict config did not load')
|
| 48 |
-
result=original(*args,**kwargs)
|
| 49 |
-
if config_fields(kernel.cache[key.autotune_key])!=config_fields(expected):raise RuntimeError('Kernel config changed during call')
|
| 50 |
-
telemetry['calls']+=1;telemetry['keys'][encoded]=telemetry['keys'].get(encoded,0)+1
|
| 51 |
-
return result
|
| 52 |
-
kernel.run=guarded
|
| 53 |
-
return original
|
| 54 |
-
|
| 55 |
-
def install(profile_path,expected_sha):
|
| 56 |
-
path,profile,entries=validate_profile(profile_path,expected_sha)
|
| 57 |
-
if any(n=='fla' or n.startswith('fla.') for n in sys.modules):raise RuntimeError('Install profile before importing FLA; use a fresh isolated process')
|
| 58 |
-
for name,wanted in {'FLA_CACHE_MODE':'strict','FLA_CONFIG_DIR':str(path.parent)}.items():
|
| 59 |
-
actual=os.environ.get(name)
|
| 60 |
-
if actual is not None and actual!=wanted:raise RuntimeError('Conflicting '+name)
|
| 61 |
-
os.environ[name]=wanted
|
| 62 |
-
import torch,triton,fla
|
| 63 |
-
actual={'torch':str(torch.__version__),'hip':torch.version.hip,'triton':triton.__version__,'fla':fla.__version__}
|
| 64 |
-
for name,value in actual.items():
|
| 65 |
-
if value!=profile['runtime'][name]:raise RuntimeError('Runtime mismatch: '+name)
|
| 66 |
-
if not torch.cuda.is_available():raise RuntimeError('Profile is only qualified for the specified ROCm GPU')
|
| 67 |
-
arch=torch.cuda.get_device_properties(0).gcnArchName.split(':')[0]
|
| 68 |
-
if arch!=profile['runtime']['gpu_arch']:raise RuntimeError('Unsupported GPU architecture '+arch)
|
| 69 |
-
module=importlib.import_module('fla.modules.l2norm');cache_module=importlib.import_module('fla.ops.utils.cache');root=Path(fla.__file__).parent
|
| 70 |
-
for name,value in profile['fla_source_sha256'].items():
|
| 71 |
-
if sha(root/name)!=value:raise RuntimeError('Pinned FLA source changed: '+name)
|
| 72 |
-
kernel=module.l2norm_fwd_kernel
|
| 73 |
-
if kernel.kernel_name!='l2norm_fwd_kernel' or kernel.keys!=['D','NB'] or kernel.cache:raise RuntimeError('Kernel identity or fresh-cache precondition failed')
|
| 74 |
-
telemetry={'profile_sha256':expected_sha,'status':'installed','calls':0,'keys':{},'strict_guard':True,'unknown_keys':'raise','autotune_fallback_permitted':False,'runtime':actual,'gpu_arch':arch,'process_scope':'one explicitly profiled Nox model; other model loading in this process is not supported'}
|
| 75 |
-
attach_guard(kernel,cache_module,entries,telemetry)
|
| 76 |
-
# The validated inference path only uses the vectorized D128 forward kernel.
|
| 77 |
-
# Other dimensions/backward must not silently enter a different autotuner.
|
| 78 |
-
for name in ['l2norm_fwd_kernel1','l2norm_bwd_kernel','l2norm_bwd_kernel1']:
|
| 79 |
-
other=getattr(module,name)
|
| 80 |
-
def reject(*args,_name=name,**kwargs):raise RuntimeError('Uncontracted normalization kernel: '+_name)
|
| 81 |
-
other.run=reject
|
| 82 |
-
return telemetry
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
code/runtime_profile.py
DELETED
|
@@ -1,36 +0,0 @@
|
|
| 1 |
-
"""Bundle-local automatic launch-profile binding, before importing FLA.
|
| 2 |
-
|
| 3 |
-
One profile is active per Python process. Repeated loading of the same verified
|
| 4 |
-
profile is allowed; mixing with an unprofiled/different-profile model is not.
|
| 5 |
-
"""
|
| 6 |
-
import hashlib,importlib.util,json,os,sys,types
|
| 7 |
-
from pathlib import Path
|
| 8 |
-
STATE='_decision_process_normalization_profile_v1'
|
| 9 |
-
def sha(p):return hashlib.sha256(Path(p).read_bytes()).hexdigest()
|
| 10 |
-
def safe(root,relative):
|
| 11 |
-
p=Path(relative)
|
| 12 |
-
if p.is_absolute() or '..' in p.parts:raise ValueError('Unsafe runtime-profile path')
|
| 13 |
-
return root/p
|
| 14 |
-
|
| 15 |
-
def ensure_profile(bundle):
|
| 16 |
-
bundle=Path(bundle).resolve();runtime=json.loads((bundle/'runtime.json').read_text());spec=runtime.get('normalization_profile');active=sys.modules.get(STATE)
|
| 17 |
-
if spec is None:
|
| 18 |
-
if active is not None:raise RuntimeError('Load unprofiled and profiled Decision models in separate processes')
|
| 19 |
-
return None
|
| 20 |
-
if spec.get('kind')!='decision-fla-l2norm-profile-v1' or spec.get('validated_arch')!='gfx942':raise ValueError('Unknown normalization profile contract')
|
| 21 |
-
profile=safe(bundle,spec['profile_file']);guard=safe(bundle,spec['guard_file'])
|
| 22 |
-
if sha(profile)!=spec['profile_sha256'] or sha(guard)!=spec['guard_sha256']:raise ValueError('Bound runtime profile bytes changed')
|
| 23 |
-
if json.loads((bundle/'decision_config.json').read_text()).get('base_model')!='Qwen/Qwen3.5-4B':raise ValueError('This bundle profile is bound to the validated Nox family only')
|
| 24 |
-
if active is not None:
|
| 25 |
-
if active.profile_sha256!=spec['profile_sha256'] or active.guard_sha256!=spec['guard_sha256']:raise RuntimeError('Different Decision normalization profile is already active; use a separate process')
|
| 26 |
-
active.guard.validate_profile(profile,spec['profile_sha256'])
|
| 27 |
-
if os.environ.get('FLA_CACHE_MODE')!='strict' or os.environ.get('FLA_CONFIG_DIR')!=active.profile_dir:raise RuntimeError('Active FLA profile environment changed')
|
| 28 |
-
return {'profile_sha256':active.profile_sha256,'guard_sha256':active.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
|
| 29 |
-
module_spec=importlib.util.spec_from_file_location('decision_profile_guard_'+spec['guard_sha256'][:16],guard);module=importlib.util.module_from_spec(module_spec);module_spec.loader.exec_module(module)
|
| 30 |
-
telemetry=module.install(profile,spec['profile_sha256'])
|
| 31 |
-
state=types.ModuleType(STATE);state.profile_sha256=spec['profile_sha256'];state.guard_sha256=spec['guard_sha256'];state.profile_dir=str(profile.parent);state.guard=module;state.telemetry=telemetry;sys.modules[STATE]=state
|
| 32 |
-
return {'profile_sha256':state.profile_sha256,'guard_sha256':state.guard_sha256,'validated_arch':'gfx942','automatic_bundle_binding':True,'scope':'single profile per process'}
|
| 33 |
-
|
| 34 |
-
def active_telemetry():
|
| 35 |
-
state=sys.modules.get(STATE)
|
| 36 |
-
return None if state is None else dict(state.telemetry)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|