Xunzhuo commited on
Commit
0bb8335
·
verified ·
1 Parent(s): bb411db

Update latest Lux comparison and54 task diagnostics

Browse files
DIAGNOSTICS.md CHANGED
@@ -6,7 +6,7 @@ These axes remain separate from headline accuracy. Probability metrics use the s
6
 
7
  | Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
8
  |---|---:|---:|---:|---:|---:|---:|
9
- | Lux-9B | 1046/1046 | 0.2937 | 0.5941 | 5.56 | 63.48 | 0.0582 |
10
  | Nox-4B | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
11
  | Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
12
  | Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
@@ -28,7 +28,7 @@ Coverage at an error threshold keeps whole confidence-tie groups together. These
28
 
29
  | Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
30
  |---|---:|---:|---:|---:|
31
- | Lux-9B | 36/36 | 77.78 | 8.33 | 0.0852 |
32
  | Nox-4B | 36/36 | 63.89 | 16.67 | 0.0795 |
33
  | Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
34
  | Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
@@ -50,7 +50,7 @@ The 36 paired permutations test the same semantics under changed option order. S
50
 
51
  | Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
52
  |---|---:|---:|---:|---:|---:|---:|
53
- | Lux-9B | 110/110 | 86.36 | 64.85 | 21.82 | 0.7185 | 25.75 |
54
  | Nox-4B | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
55
  | Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
56
  | Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
@@ -94,7 +94,7 @@ Transfer coverage includes all 1,264 questions: clean, unknown-evidence and orde
94
 
95
  | Model | Overall % | 95% component-bootstrap interval |
96
  |---|---:|---:|
97
- | Lux-9B | 77.22 | 75.83–78.59 |
98
  | Nox-4B | 73.09 | 71.57–74.56 |
99
  | Kev-9B | 71.89 | 70.42–73.35 |
100
  | Kev-4B | 70.09 | 68.45–71.63 |
 
6
 
7
  | Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
8
  |---|---:|---:|---:|---:|---:|---:|
9
+ | Lux-9B | 1046/1046 | 0.2911 | 0.5870 | 4.79 | 63.38 | 0.0579 |
10
  | Nox-4B | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
11
  | Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
12
  | Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
 
28
 
29
  | Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
30
  |---|---:|---:|---:|---:|
31
+ | Lux-9B | 36/36 | 77.78 | 8.33 | 0.0823 |
32
  | Nox-4B | 36/36 | 63.89 | 16.67 | 0.0795 |
33
  | Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
34
  | Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
 
50
 
51
  | Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
52
  |---|---:|---:|---:|---:|---:|---:|
53
+ | Lux-9B | 110/110 | 86.36 | 63.60 | 20.91 | 0.7358 | 26.59 |
54
  | Nox-4B | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
55
  | Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
56
  | Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
 
94
 
95
  | Model | Overall % | 95% component-bootstrap interval |
96
  |---|---:|---:|
97
+ | Lux-9B | 77.40 | 76.01–78.77 |
98
  | Nox-4B | 73.09 | 71.57–74.56 |
99
  | Kev-9B | 71.89 | 70.42–73.35 |
100
  | Kev-4B | 70.09 | 68.45–71.63 |
EVALUATION.md CHANGED
@@ -4,7 +4,7 @@ The comparison covers **3,766 scored decisions across 54 tasks** and all 15 disp
4
 
5
  | Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
6
  |---|---:|---:|---:|---:|---:|---:|---:|
7
- | Lux-9B | 9B | **84.17** | **52.75** | 89.69 | **91.25** | 77.63 | **77.22** |
8
  | Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
9
  | Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
10
  | Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
 
4
 
5
  | Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
6
  |---|---:|---:|---:|---:|---:|---:|---:|
7
+ | Lux-9B | 9B | **84.38** | **52.75** | 90.16 | **91.46** | 77.72 | **77.40** |
8
  | Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
9
  | Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
10
  | Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
MATERIALS.json CHANGED
@@ -3,26 +3,26 @@
3
  "public_models": 15,
4
  "scored_questions": 3766,
5
  "tasks": 54,
6
- "source_statistics_sha256": "c73006e701434d215264f2d623160f213cebf2a8ff9b4f8c0d3402b3b99436d3",
7
  "files": {
8
  ".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
9
- "DIAGNOSTICS.md": "9d9a69142f2a8c00b3bedbe5bd5ec94ba1e56a286a72dcea62af5f0b7339d5d9",
10
- "EVALUATION.md": "6cd20e879d993760e58022f060b179b8abc8465db553f327b78ab70a21483594",
11
  "QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
12
- "README.md": "5bf2ae41601b54d6cd29c0cfa43c648dea313a3420f16f78b87008184605a2dd",
13
- "SENSITIVITY.md": "b1f5899c956cd650e4bc9f6f59bac922370d36e08a8a9a5da81d54084d6a218d",
14
- "TASKS.md": "3d156b88e96e78787b20d818fee467aaa4f42eff9d741488dcaed51fe6ea2243",
15
  "USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
16
- "WEIGHTING.md": "b1f5899c956cd650e4bc9f6f59bac922370d36e08a8a9a5da81d54084d6a218d",
17
- "assets/decision-matrix.pdf": "4a5eac2379042e00876e52b425b64845e10bb66a38655ca0417bd800eaa7632a",
18
- "assets/decision-matrix.png": "cd72705657bbdb1170ca7415fadd7758225d8f13c8c3f70aa7e7aaa87b580a49",
19
- "assets/decision-matrix.svg": "0146c0ee60712dd49a76a6843283bac96b9ae06b4ac07cb4e37d870830b989e3",
20
  "assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
21
- "assets/decision-ranking.pdf": "e5ef56079533695ee36368ab3973f76c592fca42bcc02312ffbc4dcb670f5dbf",
22
- "assets/decision-ranking.png": "dd24653eec28e736178ef12927f4d3914f943542cde10e187416030f17cb6014",
23
- "assets/decision-ranking.svg": "3721021059ea15202b26b8d1869d3198018b547cb048073ec58fff8cf2f4c1e3",
24
- "metrics/benchmark.json": "ff4b098ffe69e388175c13f1883400421b03abde9fc003548dff09f6eaecaf6b",
25
- "metrics/evaluation-provenance.json": "33eb76fd165de8fe0c7a48a983222102978406270408002197db5bc6ecaf9e5c",
26
  "metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
27
  }
28
  }
 
3
  "public_models": 15,
4
  "scored_questions": 3766,
5
  "tasks": 54,
6
+ "source_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
7
  "files": {
8
  ".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
9
+ "DIAGNOSTICS.md": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa",
10
+ "EVALUATION.md": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a",
11
  "QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
12
+ "README.md": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
13
+ "SENSITIVITY.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
14
+ "TASKS.md": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247",
15
  "USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
16
+ "WEIGHTING.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
17
+ "assets/decision-matrix.pdf": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b",
18
+ "assets/decision-matrix.png": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696",
19
+ "assets/decision-matrix.svg": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc",
20
  "assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
21
+ "assets/decision-ranking.pdf": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f",
22
+ "assets/decision-ranking.png": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9",
23
+ "assets/decision-ranking.svg": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e",
24
+ "metrics/benchmark.json": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c",
25
+ "metrics/evaluation-provenance.json": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9",
26
  "metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
27
  }
28
  }
README.md CHANGED
@@ -37,7 +37,7 @@ Give Nox a state, questions and possible answers. It returns typed decisions and
37
  | Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
38
  |---|---:|---:|---:|---:|---:|---:|---:|
39
  | Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
40
- | Lux-9B | 9B | **84.17** | **52.75** | 89.69 | **91.25** | 77.63 | **77.22** |
41
  | Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
42
  | Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
43
  | Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
 
37
  | Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
38
  |---|---:|---:|---:|---:|---:|---:|---:|
39
  | Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
40
+ | Lux-9B | 9B | **84.38** | **52.75** | 90.16 | **91.46** | 77.72 | **77.40** |
41
  | Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
42
  | Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
43
  | Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
SENSITIVITY.md CHANGED
@@ -4,7 +4,7 @@ The current product-priority weights were chosen after observing results. This c
4
 
5
  | Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
6
  |---|---:|---:|---:|
7
- | Lux-9B | 77.22 | 76.90 | 79.47 |
8
  | Nox-4B | 73.09 | 72.42 | 75.03 |
9
  | Kev-9B | 71.89 | 72.01 | 73.19 |
10
  | Kev-4B | 70.09 | 70.30 | 71.73 |
 
4
 
5
  | Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
6
  |---|---:|---:|---:|
7
+ | Lux-9B | 77.40 | 77.07 | 79.69 |
8
  | Nox-4B | 73.09 | 72.42 | 75.03 |
9
  | Kev-9B | 71.89 | 72.01 | 73.19 |
10
  | Kev-4B | 70.09 | 70.30 | 71.73 |
TASKS.md CHANGED
@@ -13,9 +13,9 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
13
  | Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 89.06 | 81.25 | 81.25 | 100.00 |
14
  | Evidence placement | 96 | 72.92 | **100.00** | 50.00 | 41.67 | 79.17 | 8.33 | 72.92 | **98.96** | 19.79 | 33.33 | 9.38 | **100.00** | 95.83 | 54.17 | 30.21 |
15
  | Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 84.38 | 59.38 | 81.25 | 6.25 | 18.75 | 12.50 | 100.00 |
16
- | Relation composition | 96 | 59.38 | 52.08 | 51.04 | 62.50 | 36.46 | 54.17 | 51.04 | 37.50 | **64.58** | 44.79 | 48.96 | 40.62 | 25.00 | 35.42 | 56.25 |
17
- | Scoped evidence | 96 | **94.79** | **78.12** | 65.62 | 47.92 | 55.21 | 47.92 | 51.04 | **79.17** | 25.00 | 10.42 | 36.46 | 25.00 | 37.50 | 36.46 | 89.58 |
18
- | State tracking | 96 | 29.17 | 35.42 | 38.54 | 34.38 | 31.25 | 29.17 | 28.12 | 27.08 | 38.54 | 31.25 | 25.00 | 29.17 | 23.96 | 29.17 | 33.33 |
19
  | In / out of menu | 64 | **100.00** | **100.00** | 84.38 | 75.00 | 71.88 | 59.38 | 60.94 | **100.00** | 70.31 | 57.81 | 62.50 | 75.00 | 64.06 | 37.50 | 100.00 |
20
 
21
  </details>
@@ -26,15 +26,15 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
26
  | Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
27
  |---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
28
  | Record identity | 80 | **61.25** | 53.75 | 45.00 | 50.00 | 57.50 | 56.25 | 50.00 | 50.00 | **60.00** | 50.00 | 46.25 | 51.25 | 47.50 | 50.00 | 76.25 |
29
- | Capacity assignment | 80 | **72.50** | 46.25 | 50.00 | 50.00 | 50.00 | 51.25 | 50.00 | 47.50 | 50.00 | 50.00 | 50.00 | 50.00 | 63.75 | 50.00 | 68.75 |
30
  | Constraint assignment | 80 | 32.50 | **41.25** | 27.50 | 32.50 | 25.00 | 23.75 | 23.75 | 26.25 | 25.00 | 20.00 | 21.25 | 26.25 | 22.50 | 27.50 | 55.00 |
31
  | Intent routing · EN | 120 | 90.00 | 91.67 | 86.67 | 87.50 | 88.33 | 90.00 | 91.67 | 90.83 | 91.67 | 83.33 | 70.83 | 81.67 | 74.17 | 70.83 | 91.67 |
32
  | Intent routing · ZH | 120 | 87.50 | 87.50 | 85.83 | 86.67 | 86.67 | 88.33 | 86.67 | 87.50 | 85.00 | 85.83 | 74.17 | 79.17 | 44.17 | 73.33 | 88.33 |
33
  | Multiset reconciliation | 80 | 28.75 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 26.25 | 25.00 | 27.50 | 25.00 | 26.25 | 32.50 | 58.75 |
34
- | Ordered service loss | 80 | 27.50 | **30.00** | 25.00 | 20.00 | 20.00 | 22.50 | 21.25 | 25.00 | 20.00 | 27.50 | 20.00 | 20.00 | 20.00 | 17.50 | 42.50 |
35
  | Conflicting rule closure | 80 | **36.25** | 33.75 | 27.50 | 33.75 | 27.50 | 26.25 | 25.00 | 25.00 | 27.50 | 27.50 | 27.50 | 25.00 | 18.75 | 20.00 | 66.25 |
36
- | Temporal exclusion | 80 | 38.75 | 42.50 | 28.75 | 45.00 | 36.25 | 42.50 | 31.25 | 41.25 | 32.50 | 21.25 | 28.75 | 17.50 | 12.50 | 28.75 | 41.25 |
37
- | Transaction recovery | 80 | **52.50** | **62.50** | 43.75 | 51.25 | 35.00 | 41.25 | 26.25 | 38.75 | 42.50 | 32.50 | 23.75 | 32.50 | 23.75 | 18.75 | 75.00 |
38
 
39
  </details>
40
 
@@ -43,9 +43,9 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
43
 
44
  | Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
45
  |---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
46
- | Yes / no reading | 160 | 88.75 | 86.25 | 91.25 | 88.75 | 84.38 | 91.88 | 83.75 | 83.12 | 75.62 | 74.38 | 65.00 | 74.38 | 69.38 | 69.38 | 92.50 |
47
  | Reading · EN | 160 | 92.50 | 73.12 | 83.12 | 75.62 | 98.75 | 93.12 | 93.12 | 70.00 | 68.12 | 63.75 | 83.75 | 38.12 | 38.12 | 36.25 | 96.88 |
48
- | Reading · ZH | 160 | 88.75 | 70.62 | 81.25 | 74.38 | 91.88 | 91.25 | 91.25 | 70.00 | 61.88 | 58.75 | 81.25 | 31.87 | 28.75 | 28.12 | 96.25 |
49
 
50
  </details>
51
 
@@ -57,7 +57,7 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
57
  | Contextual reasoning | 120 | **82.50** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 70.00 | 42.50 | 56.67 | 46.67 | 30.00 | 20.00 | 86.67 |
58
  | Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 81.67 | 66.67 | 82.50 | 55.83 | 57.50 | 55.00 | 90.83 |
59
  | Textual entailment | 120 | **92.50** | 89.17 | 90.83 | 89.17 | 81.67 | 90.83 | 80.00 | 88.33 | 84.17 | 77.50 | 64.17 | 78.33 | 72.50 | 72.50 | 82.50 |
60
- | Scientific inference | 120 | 96.67 | 96.67 | 96.67 | 96.67 | 98.33 | 95.83 | 96.67 | 95.83 | 90.83 | 88.33 | 85.83 | 98.33 | 95.00 | 81.67 | 99.17 |
61
 
62
  </details>
63
 
@@ -75,10 +75,10 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
75
  | Policy negation | 32 | **96.88** | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 68.75 | 71.88 | 46.88 | 56.25 | 53.12 | 50.00 | 90.62 |
76
  | Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 50.00 | 97.50 | 62.50 | 50.00 | 67.50 | 50.00 | 100.00 |
77
  | Deadline contrast | 40 | **90.00** | 70.00 | 87.50 | 75.00 | 77.50 | 30.00 | 52.50 | 40.00 | 50.00 | 45.00 | 25.00 | 50.00 | 35.00 | 22.50 | 92.50 |
78
- | Emotion | 80 | 61.25 | 60.00 | 67.50 | 65.00 | 52.50 | 86.25 | 65.00 | 22.50 | 45.00 | 60.00 | 67.50 | 53.75 | 62.50 | 56.25 | 67.50 |
79
- | MMLU | 80 | 73.75 | 61.25 | 75.00 | 68.75 | 73.75 | 62.50 | 66.25 | 52.50 | 47.50 | 51.25 | 53.75 | 37.50 | 22.50 | 27.50 | 88.75 |
80
  | MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 17.00 | 22.50 | 28.00 | 12.50 | 11.00 | 11.50 | 84.00 |
81
- | Paraphrase | 80 | **90.00** | 85.00 | 81.25 | 81.25 | 88.75 | 77.50 | 83.75 | 80.00 | 71.25 | 58.75 | 70.00 | 47.50 | 86.25 | 75.00 | 87.50 |
82
  | Question entailment | 80 | 91.25 | 91.25 | 95.00 | 91.25 | 90.00 | 87.50 | 87.50 | 83.75 | 81.25 | 81.25 | 63.75 | 71.25 | 80.00 | 73.75 | 91.25 |
83
  | Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 97.50 | 96.25 | 96.25 | 90.00 | 90.00 | 72.50 | 100.00 |
84
  | Offensive-language detection | 80 | 75.00 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 52.50 | 72.50 | 83.75 | 78.75 | 81.25 | 82.50 | 76.25 |
 
13
  | Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 89.06 | 81.25 | 81.25 | 100.00 |
14
  | Evidence placement | 96 | 72.92 | **100.00** | 50.00 | 41.67 | 79.17 | 8.33 | 72.92 | **98.96** | 19.79 | 33.33 | 9.38 | **100.00** | 95.83 | 54.17 | 30.21 |
15
  | Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 84.38 | 59.38 | 81.25 | 6.25 | 18.75 | 12.50 | 100.00 |
16
+ | Relation composition | 96 | 60.42 | 52.08 | 51.04 | 62.50 | 36.46 | 54.17 | 51.04 | 37.50 | **64.58** | 44.79 | 48.96 | 40.62 | 25.00 | 35.42 | 56.25 |
17
+ | Scoped evidence | 96 | **96.88** | **78.12** | 65.62 | 47.92 | 55.21 | 47.92 | 51.04 | **79.17** | 25.00 | 10.42 | 36.46 | 25.00 | 37.50 | 36.46 | 89.58 |
18
+ | State tracking | 96 | 28.12 | 35.42 | 38.54 | 34.38 | 31.25 | 29.17 | 28.12 | 27.08 | 38.54 | 31.25 | 25.00 | 29.17 | 23.96 | 29.17 | 33.33 |
19
  | In / out of menu | 64 | **100.00** | **100.00** | 84.38 | 75.00 | 71.88 | 59.38 | 60.94 | **100.00** | 70.31 | 57.81 | 62.50 | 75.00 | 64.06 | 37.50 | 100.00 |
20
 
21
  </details>
 
26
  | Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
27
  |---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
28
  | Record identity | 80 | **61.25** | 53.75 | 45.00 | 50.00 | 57.50 | 56.25 | 50.00 | 50.00 | **60.00** | 50.00 | 46.25 | 51.25 | 47.50 | 50.00 | 76.25 |
29
+ | Capacity assignment | 80 | **73.75** | 46.25 | 50.00 | 50.00 | 50.00 | 51.25 | 50.00 | 47.50 | 50.00 | 50.00 | 50.00 | 50.00 | 63.75 | 50.00 | 68.75 |
30
  | Constraint assignment | 80 | 32.50 | **41.25** | 27.50 | 32.50 | 25.00 | 23.75 | 23.75 | 26.25 | 25.00 | 20.00 | 21.25 | 26.25 | 22.50 | 27.50 | 55.00 |
31
  | Intent routing · EN | 120 | 90.00 | 91.67 | 86.67 | 87.50 | 88.33 | 90.00 | 91.67 | 90.83 | 91.67 | 83.33 | 70.83 | 81.67 | 74.17 | 70.83 | 91.67 |
32
  | Intent routing · ZH | 120 | 87.50 | 87.50 | 85.83 | 86.67 | 86.67 | 88.33 | 86.67 | 87.50 | 85.00 | 85.83 | 74.17 | 79.17 | 44.17 | 73.33 | 88.33 |
33
  | Multiset reconciliation | 80 | 28.75 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 26.25 | 25.00 | 27.50 | 25.00 | 26.25 | 32.50 | 58.75 |
34
+ | Ordered service loss | 80 | 26.25 | **30.00** | 25.00 | 20.00 | 20.00 | 22.50 | 21.25 | 25.00 | 20.00 | 27.50 | 20.00 | 20.00 | 20.00 | 17.50 | 42.50 |
35
  | Conflicting rule closure | 80 | **36.25** | 33.75 | 27.50 | 33.75 | 27.50 | 26.25 | 25.00 | 25.00 | 27.50 | 27.50 | 27.50 | 25.00 | 18.75 | 20.00 | 66.25 |
36
+ | Temporal exclusion | 80 | 37.50 | 42.50 | 28.75 | 45.00 | 36.25 | 42.50 | 31.25 | 41.25 | 32.50 | 21.25 | 28.75 | 17.50 | 12.50 | 28.75 | 41.25 |
37
+ | Transaction recovery | 80 | **53.75** | **62.50** | 43.75 | 51.25 | 35.00 | 41.25 | 26.25 | 38.75 | 42.50 | 32.50 | 23.75 | 32.50 | 23.75 | 18.75 | 75.00 |
38
 
39
  </details>
40
 
 
43
 
44
  | Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
45
  |---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
46
+ | Yes / no reading | 160 | 89.38 | 86.25 | 91.25 | 88.75 | 84.38 | 91.88 | 83.75 | 83.12 | 75.62 | 74.38 | 65.00 | 74.38 | 69.38 | 69.38 | 92.50 |
47
  | Reading · EN | 160 | 92.50 | 73.12 | 83.12 | 75.62 | 98.75 | 93.12 | 93.12 | 70.00 | 68.12 | 63.75 | 83.75 | 38.12 | 38.12 | 36.25 | 96.88 |
48
+ | Reading · ZH | 160 | 89.38 | 70.62 | 81.25 | 74.38 | 91.88 | 91.25 | 91.25 | 70.00 | 61.88 | 58.75 | 81.25 | 31.87 | 28.75 | 28.12 | 96.25 |
49
 
50
  </details>
51
 
 
57
  | Contextual reasoning | 120 | **82.50** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 70.00 | 42.50 | 56.67 | 46.67 | 30.00 | 20.00 | 86.67 |
58
  | Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 81.67 | 66.67 | 82.50 | 55.83 | 57.50 | 55.00 | 90.83 |
59
  | Textual entailment | 120 | **92.50** | 89.17 | 90.83 | 89.17 | 81.67 | 90.83 | 80.00 | 88.33 | 84.17 | 77.50 | 64.17 | 78.33 | 72.50 | 72.50 | 82.50 |
60
+ | Scientific inference | 120 | 97.50 | 96.67 | 96.67 | 96.67 | 98.33 | 95.83 | 96.67 | 95.83 | 90.83 | 88.33 | 85.83 | 98.33 | 95.00 | 81.67 | 99.17 |
61
 
62
  </details>
63
 
 
75
  | Policy negation | 32 | **96.88** | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 68.75 | 71.88 | 46.88 | 56.25 | 53.12 | 50.00 | 90.62 |
76
  | Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 50.00 | 97.50 | 62.50 | 50.00 | 67.50 | 50.00 | 100.00 |
77
  | Deadline contrast | 40 | **90.00** | 70.00 | 87.50 | 75.00 | 77.50 | 30.00 | 52.50 | 40.00 | 50.00 | 45.00 | 25.00 | 50.00 | 35.00 | 22.50 | 92.50 |
78
+ | Emotion | 80 | 60.00 | 60.00 | 67.50 | 65.00 | 52.50 | 86.25 | 65.00 | 22.50 | 45.00 | 60.00 | 67.50 | 53.75 | 62.50 | 56.25 | 67.50 |
79
+ | MMLU | 80 | 75.00 | 61.25 | 75.00 | 68.75 | 73.75 | 62.50 | 66.25 | 52.50 | 47.50 | 51.25 | 53.75 | 37.50 | 22.50 | 27.50 | 88.75 |
80
  | MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 17.00 | 22.50 | 28.00 | 12.50 | 11.00 | 11.50 | 84.00 |
81
+ | Paraphrase | 80 | **91.25** | 85.00 | 81.25 | 81.25 | 88.75 | 77.50 | 83.75 | 80.00 | 71.25 | 58.75 | 70.00 | 47.50 | 86.25 | 75.00 | 87.50 |
82
  | Question entailment | 80 | 91.25 | 91.25 | 95.00 | 91.25 | 90.00 | 87.50 | 87.50 | 83.75 | 81.25 | 81.25 | 63.75 | 71.25 | 80.00 | 73.75 | 91.25 |
83
  | Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 97.50 | 96.25 | 96.25 | 90.00 | 90.00 | 72.50 | 100.00 |
84
  | Offensive-language detection | 80 | 75.00 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 52.50 | 72.50 | 83.75 | 78.75 | 81.25 | 82.50 | 76.25 |
WEIGHTING.md CHANGED
@@ -4,7 +4,7 @@ The current product-priority weights were chosen after observing results. This c
4
 
5
  | Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
6
  |---|---:|---:|---:|
7
- | Lux-9B | 77.22 | 76.90 | 79.47 |
8
  | Nox-4B | 73.09 | 72.42 | 75.03 |
9
  | Kev-9B | 71.89 | 72.01 | 73.19 |
10
  | Kev-4B | 70.09 | 70.30 | 71.73 |
 
4
 
5
  | Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
6
  |---|---:|---:|---:|
7
+ | Lux-9B | 77.40 | 77.07 | 79.69 |
8
  | Nox-4B | 73.09 | 72.42 | 75.03 |
9
  | Kev-9B | 71.89 | 72.01 | 73.19 |
10
  | Kev-4B | 70.09 | 70.30 | 71.73 |
assets/decision-matrix.pdf CHANGED
Binary files a/assets/decision-matrix.pdf and b/assets/decision-matrix.pdf differ
 
assets/decision-matrix.png CHANGED

Git LFS Details

  • SHA256: cd72705657bbdb1170ca7415fadd7758225d8f13c8c3f70aa7e7aaa87b580a49
  • Pointer size: 131 Bytes
  • Size of remote file: 376 kB

Git LFS Details

  • SHA256: cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696
  • Pointer size: 131 Bytes
  • Size of remote file: 376 kB
assets/decision-matrix.svg CHANGED
assets/decision-ranking.pdf CHANGED
Binary files a/assets/decision-ranking.pdf and b/assets/decision-ranking.pdf differ
 
assets/decision-ranking.png CHANGED

Git LFS Details

  • SHA256: dd24653eec28e736178ef12927f4d3914f943542cde10e187416030f17cb6014
  • Pointer size: 131 Bytes
  • Size of remote file: 248 kB

Git LFS Details

  • SHA256: 9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9
  • Pointer size: 131 Bytes
  • Size of remote file: 248 kB
assets/decision-ranking.svg CHANGED
metrics/benchmark.json CHANGED
The diff for this file is too large to render. See raw diff
 
metrics/evaluation-provenance.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
  "frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
3
- "merged_statistics_sha256": "c73006e701434d215264f2d623160f213cebf2a8ff9b4f8c0d3402b3b99436d3",
4
  "source_per_model": {
5
  "Jev": {
6
  "statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
@@ -111,9 +111,9 @@
111
  }
112
  },
113
  "Lux": {
114
- "statistics_sha256": "49c1195fcd88860620fa7d5eb48bb77d6bc759cbac6db89ab11a6faf485644fa",
115
- "source_model_key": "Lux-semantic-format-candidate",
116
- "manifest_sha256": "485e1f7505d390fb6e84c34edca5812c6803f4d7a48b3f752f7784decdf58559",
117
  "qualified_runtime": {
118
  "python": "3.12.13",
119
  "numpy": "2.3.5"
@@ -157,27 +157,27 @@
157
  "shared_reference_rows_exact": [
158
  [
159
  "Nox",
160
- "Qwen3.5-9B"
161
  ],
162
  [
163
  "Nox",
164
- "Qwen3.5-4B"
165
  ],
166
  [
167
  "Nox",
168
- "llm2jev-2b"
169
  ],
170
  [
171
  "Nox",
172
- "Qwen3.5-2B"
173
  ],
174
  [
175
  "Nox",
176
- "Decider"
177
  ],
178
  [
179
  "Nox",
180
- "Laya-multilingual"
181
  ],
182
  [
183
  "Nox",
@@ -185,59 +185,59 @@
185
  ],
186
  [
187
  "Nox",
188
- "nimble-9b"
189
  ],
190
  [
191
  "Nox",
192
- "kev-4b"
193
  ],
194
  [
195
  "Nox",
196
- "Nox-null-description-candidate"
197
  ],
198
  [
199
  "Nox",
200
- "kev-9b"
201
  ],
202
  [
203
  "Nox",
204
- "Sol"
205
  ],
206
  [
207
  "Nox",
208
- "Laya-base"
209
  ],
210
  [
211
  "Nox",
212
- "llm2jev-4b"
213
  ],
214
  [
215
  "Nox",
216
- "Jev"
217
  ],
218
  [
219
  "Lux",
220
- "Qwen3.5-9B"
221
  ],
222
  [
223
  "Lux",
224
- "Qwen3.5-4B"
225
  ],
226
  [
227
  "Lux",
228
- "llm2jev-2b"
229
  ],
230
  [
231
  "Lux",
232
- "Qwen3.5-2B"
233
  ],
234
  [
235
  "Lux",
236
- "Decider"
237
  ],
238
  [
239
  "Lux",
240
- "Laya-multilingual"
241
  ],
242
  [
243
  "Lux",
@@ -245,67 +245,67 @@
245
  ],
246
  [
247
  "Lux",
248
- "nimble-9b"
249
  ],
250
  [
251
  "Lux",
252
- "kev-4b"
253
  ],
254
  [
255
  "Lux",
256
- "Nox-null-description-candidate"
257
  ],
258
  [
259
  "Lux",
260
- "kev-9b"
261
  ],
262
  [
263
  "Lux",
264
- "Sol"
265
  ],
266
  [
267
  "Lux",
268
- "Laya-base"
269
  ],
270
  [
271
  "Lux",
272
- "llm2jev-4b"
273
  ],
274
  [
275
  "Lux",
276
- "Jev"
277
  ],
278
  [
279
  "Kai",
280
- "Qwen3.5-2B"
281
  ],
282
  [
283
  "Kai",
284
- "kev-0.8b"
285
  ],
286
  [
287
  "Kai",
288
- "Laya-base"
289
  ],
290
  [
291
  "Kai",
292
- "Laya-multilingual"
293
  ],
294
  [
295
  "Eos",
296
- "Qwen3.5-2B"
297
  ],
298
  [
299
  "Eos",
300
- "kev-0.8b"
301
  ],
302
  [
303
  "Eos",
304
- "Laya-multilingual"
305
  ],
306
  [
307
  "Eos",
308
- "Laya-base"
309
  ]
310
  ],
311
  "published_matrices_exact": {
@@ -319,5 +319,5 @@
319
  }
320
  }
321
  },
322
- "source_builder_sha256": "1aacc47ba4f54768552a238a60e78736d9a54467f5eb6d955d2a0a4e0149c095"
323
  }
 
1
  {
2
  "frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
3
+ "merged_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
4
  "source_per_model": {
5
  "Jev": {
6
  "statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
 
111
  }
112
  },
113
  "Lux": {
114
+ "statistics_sha256": "71251a31ed6435d8c94df6d2a5241e9183613c19a87f8970145afec8ec948efc",
115
+ "source_model_key": "Lux-reading-source-candidate",
116
+ "manifest_sha256": "7107a2a8264a0416fa75fe40e57f3d06e4cb0395527eaedfa7b7335d8e37109a",
117
  "qualified_runtime": {
118
  "python": "3.12.13",
119
  "numpy": "2.3.5"
 
157
  "shared_reference_rows_exact": [
158
  [
159
  "Nox",
160
+ "llm2jev-2b"
161
  ],
162
  [
163
  "Nox",
164
+ "kev-9b"
165
  ],
166
  [
167
  "Nox",
168
+ "Sol"
169
  ],
170
  [
171
  "Nox",
172
+ "Nox-null-description-candidate"
173
  ],
174
  [
175
  "Nox",
176
+ "Qwen3.5-2B"
177
  ],
178
  [
179
  "Nox",
180
+ "Qwen3.5-9B"
181
  ],
182
  [
183
  "Nox",
 
185
  ],
186
  [
187
  "Nox",
188
+ "kev-4b"
189
  ],
190
  [
191
  "Nox",
192
+ "Decider"
193
  ],
194
  [
195
  "Nox",
196
+ "Laya-base"
197
  ],
198
  [
199
  "Nox",
200
+ "Jev"
201
  ],
202
  [
203
  "Nox",
204
+ "Laya-multilingual"
205
  ],
206
  [
207
  "Nox",
208
+ "llm2jev-4b"
209
  ],
210
  [
211
  "Nox",
212
+ "nimble-9b"
213
  ],
214
  [
215
  "Nox",
216
+ "Qwen3.5-4B"
217
  ],
218
  [
219
  "Lux",
220
+ "llm2jev-2b"
221
  ],
222
  [
223
  "Lux",
224
+ "kev-9b"
225
  ],
226
  [
227
  "Lux",
228
+ "Sol"
229
  ],
230
  [
231
  "Lux",
232
+ "Nox-null-description-candidate"
233
  ],
234
  [
235
  "Lux",
236
+ "Qwen3.5-2B"
237
  ],
238
  [
239
  "Lux",
240
+ "Qwen3.5-9B"
241
  ],
242
  [
243
  "Lux",
 
245
  ],
246
  [
247
  "Lux",
248
+ "kev-4b"
249
  ],
250
  [
251
  "Lux",
252
+ "Decider"
253
  ],
254
  [
255
  "Lux",
256
+ "Laya-base"
257
  ],
258
  [
259
  "Lux",
260
+ "Jev"
261
  ],
262
  [
263
  "Lux",
264
+ "Laya-multilingual"
265
  ],
266
  [
267
  "Lux",
268
+ "llm2jev-4b"
269
  ],
270
  [
271
  "Lux",
272
+ "nimble-9b"
273
  ],
274
  [
275
  "Lux",
276
+ "Qwen3.5-4B"
277
  ],
278
  [
279
  "Kai",
280
+ "Laya-base"
281
  ],
282
  [
283
  "Kai",
284
+ "Laya-multilingual"
285
  ],
286
  [
287
  "Kai",
288
+ "kev-0.8b"
289
  ],
290
  [
291
  "Kai",
292
+ "Qwen3.5-2B"
293
  ],
294
  [
295
  "Eos",
296
+ "Laya-base"
297
  ],
298
  [
299
  "Eos",
300
+ "Laya-multilingual"
301
  ],
302
  [
303
  "Eos",
304
+ "kev-0.8b"
305
  ],
306
  [
307
  "Eos",
308
+ "Qwen3.5-2B"
309
  ]
310
  ],
311
  "published_matrices_exact": {
 
319
  }
320
  }
321
  },
322
+ "source_builder_sha256": "6d7aab8b701df4fe67750a2e0212fcc1e6321b4dad71d5723b3a2d6fd97e1ecd"
323
  }
release-manifest.json CHANGED
@@ -3,7 +3,7 @@
3
  "status": "qualified-runtime-and-current-documents-assembled",
4
  "bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
5
  "readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
6
- "model_card_sha256": "5bf2ae41601b54d6cd29c0cfa43c648dea313a3420f16f78b87008184605a2dd",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
8
  "assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
9
  "original_bundle_manifest_preserved": false,
@@ -22,7 +22,7 @@
22
  {
23
  "file": "DIAGNOSTICS.md",
24
  "bytes": 6327,
25
- "sha256": "9d9a69142f2a8c00b3bedbe5bd5ec94ba1e56a286a72dcea62af5f0b7339d5d9"
26
  },
27
  {
28
  "file": "Dockerfile.runtime",
@@ -32,7 +32,7 @@
32
  {
33
  "file": "EVALUATION.md",
34
  "bytes": 4079,
35
- "sha256": "6cd20e879d993760e58022f060b179b8abc8465db553f327b78ab70a21483594"
36
  },
37
  {
38
  "file": "LICENSE",
@@ -42,7 +42,7 @@
42
  {
43
  "file": "MATERIALS.json",
44
  "bytes": 2064,
45
- "sha256": "ef3c540dee8d31f8a221b5c04b1c62c294aee5d63fef4aef54bff531f8cd92c4"
46
  },
47
  {
48
  "file": "NORMALIZATION_RUNTIME.md",
@@ -67,7 +67,7 @@
67
  {
68
  "file": "README.md",
69
  "bytes": 6169,
70
- "sha256": "5bf2ae41601b54d6cd29c0cfa43c648dea313a3420f16f78b87008184605a2dd"
71
  },
72
  {
73
  "file": "RUNTIME-RELEASE.json",
@@ -87,7 +87,7 @@
87
  {
88
  "file": "SENSITIVITY.md",
89
  "bytes": 866,
90
- "sha256": "b1f5899c956cd650e4bc9f6f59bac922370d36e08a8a9a5da81d54084d6a218d"
91
  },
92
  {
93
  "file": "SERVING_OPTIMIZATION.json",
@@ -102,7 +102,7 @@
102
  {
103
  "file": "TASKS.md",
104
  "bytes": 10393,
105
- "sha256": "3d156b88e96e78787b20d818fee467aaa4f42eff9d741488dcaed51fe6ea2243"
106
  },
107
  {
108
  "file": "USAGE.md",
@@ -112,7 +112,7 @@
112
  {
113
  "file": "WEIGHTING.md",
114
  "bytes": 866,
115
- "sha256": "b1f5899c956cd650e4bc9f6f59bac922370d36e08a8a9a5da81d54084d6a218d"
116
  },
117
  {
118
  "file": "assets/architecture.png",
@@ -251,18 +251,18 @@
251
  },
252
  {
253
  "file": "assets/decision-matrix.pdf",
254
- "bytes": 29431,
255
- "sha256": "4a5eac2379042e00876e52b425b64845e10bb66a38655ca0417bd800eaa7632a"
256
  },
257
  {
258
  "file": "assets/decision-matrix.png",
259
- "bytes": 376232,
260
- "sha256": "cd72705657bbdb1170ca7415fadd7758225d8f13c8c3f70aa7e7aaa87b580a49"
261
  },
262
  {
263
  "file": "assets/decision-matrix.svg",
264
  "bytes": 50107,
265
- "sha256": "0146c0ee60712dd49a76a6843283bac96b9ae06b4ac07cb4e37d870830b989e3"
266
  },
267
  {
268
  "file": "assets/decision-nox-4b-header.png",
@@ -292,17 +292,17 @@
292
  {
293
  "file": "assets/decision-ranking.pdf",
294
  "bytes": 25376,
295
- "sha256": "e5ef56079533695ee36368ab3973f76c592fca42bcc02312ffbc4dcb670f5dbf"
296
  },
297
  {
298
  "file": "assets/decision-ranking.png",
299
- "bytes": 248399,
300
- "sha256": "dd24653eec28e736178ef12927f4d3914f943542cde10e187416030f17cb6014"
301
  },
302
  {
303
  "file": "assets/decision-ranking.svg",
304
  "bytes": 15728,
305
- "sha256": "3721021059ea15202b26b8d1869d3198018b547cb048073ec58fff8cf2f4c1e3"
306
  },
307
  {
308
  "file": "assets/readout.png",
@@ -381,8 +381,8 @@
381
  },
382
  {
383
  "file": "metrics/benchmark.json",
384
- "bytes": 2569521,
385
- "sha256": "ff4b098ffe69e388175c13f1883400421b03abde9fc003548dff09f6eaecaf6b"
386
  },
387
  {
388
  "file": "metrics/comparator-coverage.json",
@@ -391,8 +391,8 @@
391
  },
392
  {
393
  "file": "metrics/evaluation-provenance.json",
394
- "bytes": 8511,
395
- "sha256": "33eb76fd165de8fe0c7a48a983222102978406270408002197db5bc6ecaf9e5c"
396
  },
397
  {
398
  "file": "metrics/expanded-quality.json",
@@ -580,9 +580,9 @@
580
  "scope": "Latest qualified Lux comparison and54 task diagnostics; all other model rows and inference files unchanged.",
581
  "release_tag": "v1.3.2",
582
  "change_kind": "latest-comparison-documents-only",
583
- "previous_main_revision": "65a7eb4c103c171bb12690955640905c4df05e18",
584
- "previous_release_manifest_sha256": "c06b1c8c473239be38d9af3d9d6614f4cffa2cc8697f0a4ff36f4c7f8ddbd961",
585
- "presentation_amendment_sha256": "67216f8f7e1d8d3489859e7517fa1fb564ca80443039abaa8ae61d2bd2ff46f0",
586
- "statistics_sha256": "c73006e701434d215264f2d623160f213cebf2a8ff9b4f8c0d3402b3b99436d3",
587
  "latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
588
  }
 
3
  "status": "qualified-runtime-and-current-documents-assembled",
4
  "bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
5
  "readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
6
+ "model_card_sha256": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
7
  "repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
8
  "assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
9
  "original_bundle_manifest_preserved": false,
 
22
  {
23
  "file": "DIAGNOSTICS.md",
24
  "bytes": 6327,
25
+ "sha256": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa"
26
  },
27
  {
28
  "file": "Dockerfile.runtime",
 
32
  {
33
  "file": "EVALUATION.md",
34
  "bytes": 4079,
35
+ "sha256": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a"
36
  },
37
  {
38
  "file": "LICENSE",
 
42
  {
43
  "file": "MATERIALS.json",
44
  "bytes": 2064,
45
+ "sha256": "483718bbc312bfc1311db9900d7767d5a8bb9ac69fb87edf89a95d4d15b80eea"
46
  },
47
  {
48
  "file": "NORMALIZATION_RUNTIME.md",
 
67
  {
68
  "file": "README.md",
69
  "bytes": 6169,
70
+ "sha256": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5"
71
  },
72
  {
73
  "file": "RUNTIME-RELEASE.json",
 
87
  {
88
  "file": "SENSITIVITY.md",
89
  "bytes": 866,
90
+ "sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5"
91
  },
92
  {
93
  "file": "SERVING_OPTIMIZATION.json",
 
102
  {
103
  "file": "TASKS.md",
104
  "bytes": 10393,
105
+ "sha256": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247"
106
  },
107
  {
108
  "file": "USAGE.md",
 
112
  {
113
  "file": "WEIGHTING.md",
114
  "bytes": 866,
115
+ "sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5"
116
  },
117
  {
118
  "file": "assets/architecture.png",
 
251
  },
252
  {
253
  "file": "assets/decision-matrix.pdf",
254
+ "bytes": 29410,
255
+ "sha256": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b"
256
  },
257
  {
258
  "file": "assets/decision-matrix.png",
259
+ "bytes": 375860,
260
+ "sha256": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696"
261
  },
262
  {
263
  "file": "assets/decision-matrix.svg",
264
  "bytes": 50107,
265
+ "sha256": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc"
266
  },
267
  {
268
  "file": "assets/decision-nox-4b-header.png",
 
292
  {
293
  "file": "assets/decision-ranking.pdf",
294
  "bytes": 25376,
295
+ "sha256": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f"
296
  },
297
  {
298
  "file": "assets/decision-ranking.png",
299
+ "bytes": 248285,
300
+ "sha256": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9"
301
  },
302
  {
303
  "file": "assets/decision-ranking.svg",
304
  "bytes": 15728,
305
+ "sha256": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e"
306
  },
307
  {
308
  "file": "assets/readout.png",
 
381
  },
382
  {
383
  "file": "metrics/benchmark.json",
384
+ "bytes": 2569391,
385
+ "sha256": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c"
386
  },
387
  {
388
  "file": "metrics/comparator-coverage.json",
 
391
  },
392
  {
393
  "file": "metrics/evaluation-provenance.json",
394
+ "bytes": 8510,
395
+ "sha256": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9"
396
  },
397
  {
398
  "file": "metrics/expanded-quality.json",
 
580
  "scope": "Latest qualified Lux comparison and54 task diagnostics; all other model rows and inference files unchanged.",
581
  "release_tag": "v1.3.2",
582
  "change_kind": "latest-comparison-documents-only",
583
+ "previous_main_revision": "bb411db257ca34a99a085764fcf44d3a7c7c2874",
584
+ "previous_release_manifest_sha256": "8063ae6e1048516567577001f922d1ed6abf8701f5fc4888166330c9bedbaa40",
585
+ "presentation_amendment_sha256": "4fadc67b276d2b1e79e19ccacd383482cb0580996450d0103ffe2a90cb494a56",
586
+ "statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
587
  "latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
588
  }