Update latest Lux comparison and54 task diagnostics
Browse files- DIAGNOSTICS.md +4 -4
- EVALUATION.md +1 -1
- MATERIALS.json +15 -15
- README.md +1 -1
- SENSITIVITY.md +1 -1
- TASKS.md +13 -13
- WEIGHTING.md +1 -1
- assets/decision-matrix.pdf +0 -0
- assets/decision-matrix.png +2 -2
- assets/decision-matrix.svg +127 -127
- assets/decision-ranking.pdf +0 -0
- assets/decision-ranking.png +2 -2
- assets/decision-ranking.svg +25 -25
- metrics/benchmark.json +0 -0
- metrics/evaluation-provenance.json +41 -41
- release-manifest.json +25 -25
DIAGNOSTICS.md
CHANGED
|
@@ -6,7 +6,7 @@ These axes remain separate from headline accuracy. Probability metrics use the s
|
|
| 6 |
|
| 7 |
| Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
|
| 8 |
|---|---:|---:|---:|---:|---:|---:|
|
| 9 |
-
| Lux-9B | 1046/1046 | 0.
|
| 10 |
| Nox-4B | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
|
| 11 |
| Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
|
| 12 |
| Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
|
|
@@ -28,7 +28,7 @@ Coverage at an error threshold keeps whole confidence-tie groups together. These
|
|
| 28 |
|
| 29 |
| Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
|
| 30 |
|---|---:|---:|---:|---:|
|
| 31 |
-
| Lux-9B | 36/36 | 77.78 | 8.33 | 0.
|
| 32 |
| Nox-4B | 36/36 | 63.89 | 16.67 | 0.0795 |
|
| 33 |
| Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
|
| 34 |
| Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
|
|
@@ -50,7 +50,7 @@ The 36 paired permutations test the same semantics under changed option order. S
|
|
| 50 |
|
| 51 |
| Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
|
| 52 |
|---|---:|---:|---:|---:|---:|---:|
|
| 53 |
-
| Lux-9B | 110/110 | 86.36 |
|
| 54 |
| Nox-4B | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
|
| 55 |
| Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
|
| 56 |
| Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
|
|
@@ -94,7 +94,7 @@ Transfer coverage includes all 1,264 questions: clean, unknown-evidence and orde
|
|
| 94 |
|
| 95 |
| Model | Overall % | 95% component-bootstrap interval |
|
| 96 |
|---|---:|---:|
|
| 97 |
-
| Lux-9B | 77.
|
| 98 |
| Nox-4B | 73.09 | 71.57–74.56 |
|
| 99 |
| Kev-9B | 71.89 | 70.42–73.35 |
|
| 100 |
| Kev-4B | 70.09 | 68.45–71.63 |
|
|
|
|
| 6 |
|
| 7 |
| Model | Valid / requested | Brier ↓ | NLL ↓ | ECE % ↓ | Coverage at ≤5% error % ↑ | AURC ↓ |
|
| 8 |
|---|---:|---:|---:|---:|---:|---:|
|
| 9 |
+
| Lux-9B | 1046/1046 | 0.2911 | 0.5870 | 4.79 | 63.38 | 0.0579 |
|
| 10 |
| Nox-4B | 1046/1046 | 0.4169 | 0.9079 | 11.80 | 36.42 | 0.1109 |
|
| 11 |
| Kev-9B | 1046/1046 | 0.2948 | 0.6028 | 4.45 | 58.03 | 0.0586 |
|
| 12 |
| Kev-4B | 1046/1046 | 0.3145 | 0.6434 | 2.61 | 56.79 | 0.0659 |
|
|
|
|
| 28 |
|
| 29 |
| Model | Valid / requested pairs | Both correct % ↑ | Semantic flip % ↓ | Mean half-L1 ↓ |
|
| 30 |
|---|---:|---:|---:|---:|
|
| 31 |
+
| Lux-9B | 36/36 | 77.78 | 8.33 | 0.0823 |
|
| 32 |
| Nox-4B | 36/36 | 63.89 | 16.67 | 0.0795 |
|
| 33 |
| Kev-9B | 36/36 | 80.56 | 2.78 | 0.0615 |
|
| 34 |
| Kev-4B | 36/36 | 77.78 | 5.56 | 0.0667 |
|
|
|
|
| 50 |
|
| 51 |
| Model | Valid / requested | Intact/control accuracy % ↑ | Mean max P % ↓ | P≥0.9 share % ↓ | Normalized entropy ↑ | Paired confidence drop pp ↑ |
|
| 52 |
|---|---:|---:|---:|---:|---:|---:|
|
| 53 |
+
| Lux-9B | 110/110 | 86.36 | 63.60 | 20.91 | 0.7358 | 26.59 |
|
| 54 |
| Nox-4B | 110/110 | 72.73 | 78.65 | 27.27 | 0.5072 | 12.64 |
|
| 55 |
| Kev-9B | 110/110 | 91.82 | 39.61 | 0.00 | 0.9981 | 53.19 |
|
| 56 |
| Kev-4B | 110/110 | 91.82 | 41.47 | 0.00 | 0.9918 | 51.57 |
|
|
|
|
| 94 |
|
| 95 |
| Model | Overall % | 95% component-bootstrap interval |
|
| 96 |
|---|---:|---:|
|
| 97 |
+
| Lux-9B | 77.40 | 76.01–78.77 |
|
| 98 |
| Nox-4B | 73.09 | 71.57–74.56 |
|
| 99 |
| Kev-9B | 71.89 | 70.42–73.35 |
|
| 100 |
| Kev-4B | 70.09 | 68.45–71.63 |
|
EVALUATION.md
CHANGED
|
@@ -4,7 +4,7 @@ The comparison covers **3,766 scored decisions across 54 tasks** and all 15 disp
|
|
| 4 |
|
| 5 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 6 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 7 |
-
| Lux-9B | 9B | **84.
|
| 8 |
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 9 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 10 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
|
|
|
| 4 |
|
| 5 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 6 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 9B | **84.38** | **52.75** | 90.16 | **91.46** | 77.72 | **77.40** |
|
| 8 |
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 9 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 10 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
MATERIALS.json
CHANGED
|
@@ -3,26 +3,26 @@
|
|
| 3 |
"public_models": 15,
|
| 4 |
"scored_questions": 3766,
|
| 5 |
"tasks": 54,
|
| 6 |
-
"source_statistics_sha256": "
|
| 7 |
"files": {
|
| 8 |
".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
|
| 9 |
-
"DIAGNOSTICS.md": "
|
| 10 |
-
"EVALUATION.md": "
|
| 11 |
"QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
|
| 12 |
-
"README.md": "
|
| 13 |
-
"SENSITIVITY.md": "
|
| 14 |
-
"TASKS.md": "
|
| 15 |
"USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
|
| 16 |
-
"WEIGHTING.md": "
|
| 17 |
-
"assets/decision-matrix.pdf": "
|
| 18 |
-
"assets/decision-matrix.png": "
|
| 19 |
-
"assets/decision-matrix.svg": "
|
| 20 |
"assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
|
| 21 |
-
"assets/decision-ranking.pdf": "
|
| 22 |
-
"assets/decision-ranking.png": "
|
| 23 |
-
"assets/decision-ranking.svg": "
|
| 24 |
-
"metrics/benchmark.json": "
|
| 25 |
-
"metrics/evaluation-provenance.json": "
|
| 26 |
"metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
|
| 27 |
}
|
| 28 |
}
|
|
|
|
| 3 |
"public_models": 15,
|
| 4 |
"scored_questions": 3766,
|
| 5 |
"tasks": 54,
|
| 6 |
+
"source_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
|
| 7 |
"files": {
|
| 8 |
".gitattributes": "f0cd3e623808977834bdd29b1ac3258f54a5d581affa46a7e7ab8587da26cdd9",
|
| 9 |
+
"DIAGNOSTICS.md": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa",
|
| 10 |
+
"EVALUATION.md": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a",
|
| 11 |
"QUESTION-SCALING.md": "02e8d66f6c6dc2a38f39748247c521e40648d2b673eba95f7be78ee729e929d7",
|
| 12 |
+
"README.md": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
|
| 13 |
+
"SENSITIVITY.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
|
| 14 |
+
"TASKS.md": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247",
|
| 15 |
"USAGE.md": "a5474ee65259f977ee0410a1d1d35a9662cb904c7953bef44fbf39a883361721",
|
| 16 |
+
"WEIGHTING.md": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5",
|
| 17 |
+
"assets/decision-matrix.pdf": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b",
|
| 18 |
+
"assets/decision-matrix.png": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696",
|
| 19 |
+
"assets/decision-matrix.svg": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc",
|
| 20 |
"assets/decision-nox-4b-header.png": "c79b9a7b122e7e72485095ed28ba2ff515a81e1ab49a778654b0cddbe2ecafac",
|
| 21 |
+
"assets/decision-ranking.pdf": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f",
|
| 22 |
+
"assets/decision-ranking.png": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9",
|
| 23 |
+
"assets/decision-ranking.svg": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e",
|
| 24 |
+
"metrics/benchmark.json": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c",
|
| 25 |
+
"metrics/evaluation-provenance.json": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9",
|
| 26 |
"metrics/question-scaling.json": "3aaab58d54b78ccfd08b74f7f45697eee63a6d45ca9819d22da3ee9d437d47b8"
|
| 27 |
}
|
| 28 |
}
|
README.md
CHANGED
|
@@ -37,7 +37,7 @@ Give Nox a state, questions and possible answers. It returns typed decisions and
|
|
| 37 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 38 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 39 |
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 40 |
-
| Lux-9B | 9B | **84.
|
| 41 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 42 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 43 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
|
|
|
| 37 |
| Model | Size | Decisions | Composition | Reading | Inference | Transfer | Overall |
|
| 38 |
|---|---:|---:|---:|---:|---:|---:|---:|
|
| 39 |
| Nox-4B | 4B | **83.00** | **51.79** | 79.06 | **86.25** | 69.60 | **73.09** |
|
| 40 |
+
| Lux-9B | 9B | **84.38** | **52.75** | 90.16 | **91.46** | 77.72 | **77.40** |
|
| 41 |
| Kev-9B | 9B | 76.75 | 45.75 | 86.72 | 83.54 | 79.25 | 71.89 |
|
| 42 |
| Kev-4B | 4B | 71.90 | 48.54 | 81.88 | 84.58 | 76.10 | 70.09 |
|
| 43 |
| Qwen3.5-9B | 9B | 73.91 | 44.62 | 89.84 | 79.58 | 73.23 | 69.73 |
|
SENSITIVITY.md
CHANGED
|
@@ -4,7 +4,7 @@ The current product-priority weights were chosen after observing results. This c
|
|
| 4 |
|
| 5 |
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
-
| Lux-9B | 77.
|
| 8 |
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
|
|
|
| 4 |
|
| 5 |
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 77.40 | 77.07 | 79.69 |
|
| 8 |
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
TASKS.md
CHANGED
|
@@ -13,9 +13,9 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 13 |
| Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 89.06 | 81.25 | 81.25 | 100.00 |
|
| 14 |
| Evidence placement | 96 | 72.92 | **100.00** | 50.00 | 41.67 | 79.17 | 8.33 | 72.92 | **98.96** | 19.79 | 33.33 | 9.38 | **100.00** | 95.83 | 54.17 | 30.21 |
|
| 15 |
| Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 84.38 | 59.38 | 81.25 | 6.25 | 18.75 | 12.50 | 100.00 |
|
| 16 |
-
| Relation composition | 96 |
|
| 17 |
-
| Scoped evidence | 96 | **
|
| 18 |
-
| State tracking | 96 |
|
| 19 |
| In / out of menu | 64 | **100.00** | **100.00** | 84.38 | 75.00 | 71.88 | 59.38 | 60.94 | **100.00** | 70.31 | 57.81 | 62.50 | 75.00 | 64.06 | 37.50 | 100.00 |
|
| 20 |
|
| 21 |
</details>
|
|
@@ -26,15 +26,15 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 26 |
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 27 |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 28 |
| Record identity | 80 | **61.25** | 53.75 | 45.00 | 50.00 | 57.50 | 56.25 | 50.00 | 50.00 | **60.00** | 50.00 | 46.25 | 51.25 | 47.50 | 50.00 | 76.25 |
|
| 29 |
-
| Capacity assignment | 80 | **
|
| 30 |
| Constraint assignment | 80 | 32.50 | **41.25** | 27.50 | 32.50 | 25.00 | 23.75 | 23.75 | 26.25 | 25.00 | 20.00 | 21.25 | 26.25 | 22.50 | 27.50 | 55.00 |
|
| 31 |
| Intent routing · EN | 120 | 90.00 | 91.67 | 86.67 | 87.50 | 88.33 | 90.00 | 91.67 | 90.83 | 91.67 | 83.33 | 70.83 | 81.67 | 74.17 | 70.83 | 91.67 |
|
| 32 |
| Intent routing · ZH | 120 | 87.50 | 87.50 | 85.83 | 86.67 | 86.67 | 88.33 | 86.67 | 87.50 | 85.00 | 85.83 | 74.17 | 79.17 | 44.17 | 73.33 | 88.33 |
|
| 33 |
| Multiset reconciliation | 80 | 28.75 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 26.25 | 25.00 | 27.50 | 25.00 | 26.25 | 32.50 | 58.75 |
|
| 34 |
-
| Ordered service loss | 80 |
|
| 35 |
| Conflicting rule closure | 80 | **36.25** | 33.75 | 27.50 | 33.75 | 27.50 | 26.25 | 25.00 | 25.00 | 27.50 | 27.50 | 27.50 | 25.00 | 18.75 | 20.00 | 66.25 |
|
| 36 |
-
| Temporal exclusion | 80 |
|
| 37 |
-
| Transaction recovery | 80 | **
|
| 38 |
|
| 39 |
</details>
|
| 40 |
|
|
@@ -43,9 +43,9 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 43 |
|
| 44 |
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 45 |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 46 |
-
| Yes / no reading | 160 |
|
| 47 |
| Reading · EN | 160 | 92.50 | 73.12 | 83.12 | 75.62 | 98.75 | 93.12 | 93.12 | 70.00 | 68.12 | 63.75 | 83.75 | 38.12 | 38.12 | 36.25 | 96.88 |
|
| 48 |
-
| Reading · ZH | 160 |
|
| 49 |
|
| 50 |
</details>
|
| 51 |
|
|
@@ -57,7 +57,7 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 57 |
| Contextual reasoning | 120 | **82.50** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 70.00 | 42.50 | 56.67 | 46.67 | 30.00 | 20.00 | 86.67 |
|
| 58 |
| Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 81.67 | 66.67 | 82.50 | 55.83 | 57.50 | 55.00 | 90.83 |
|
| 59 |
| Textual entailment | 120 | **92.50** | 89.17 | 90.83 | 89.17 | 81.67 | 90.83 | 80.00 | 88.33 | 84.17 | 77.50 | 64.17 | 78.33 | 72.50 | 72.50 | 82.50 |
|
| 60 |
-
| Scientific inference | 120 |
|
| 61 |
|
| 62 |
</details>
|
| 63 |
|
|
@@ -75,10 +75,10 @@ Accuracy (%) on the same requested rows. Bold marks a Decision-family result str
|
|
| 75 |
| Policy negation | 32 | **96.88** | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 68.75 | 71.88 | 46.88 | 56.25 | 53.12 | 50.00 | 90.62 |
|
| 76 |
| Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 50.00 | 97.50 | 62.50 | 50.00 | 67.50 | 50.00 | 100.00 |
|
| 77 |
| Deadline contrast | 40 | **90.00** | 70.00 | 87.50 | 75.00 | 77.50 | 30.00 | 52.50 | 40.00 | 50.00 | 45.00 | 25.00 | 50.00 | 35.00 | 22.50 | 92.50 |
|
| 78 |
-
| Emotion | 80 |
|
| 79 |
-
| MMLU | 80 |
|
| 80 |
| MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 17.00 | 22.50 | 28.00 | 12.50 | 11.00 | 11.50 | 84.00 |
|
| 81 |
-
| Paraphrase | 80 | **
|
| 82 |
| Question entailment | 80 | 91.25 | 91.25 | 95.00 | 91.25 | 90.00 | 87.50 | 87.50 | 83.75 | 81.25 | 81.25 | 63.75 | 71.25 | 80.00 | 73.75 | 91.25 |
|
| 83 |
| Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 97.50 | 96.25 | 96.25 | 90.00 | 90.00 | 72.50 | 100.00 |
|
| 84 |
| Offensive-language detection | 80 | 75.00 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 52.50 | 72.50 | 83.75 | 78.75 | 81.25 | 82.50 | 76.25 |
|
|
|
|
| 13 |
| Intent routing | 64 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 100.00 | 96.88 | 89.06 | 81.25 | 81.25 | 100.00 |
|
| 14 |
| Evidence placement | 96 | 72.92 | **100.00** | 50.00 | 41.67 | 79.17 | 8.33 | 72.92 | **98.96** | 19.79 | 33.33 | 9.38 | **100.00** | 95.83 | 54.17 | 30.21 |
|
| 15 |
| Ordered rubric | 64 | 100.00 | 89.06 | 96.88 | 100.00 | 84.38 | 84.38 | 90.62 | 65.62 | 84.38 | 59.38 | 81.25 | 6.25 | 18.75 | 12.50 | 100.00 |
|
| 16 |
+
| Relation composition | 96 | 60.42 | 52.08 | 51.04 | 62.50 | 36.46 | 54.17 | 51.04 | 37.50 | **64.58** | 44.79 | 48.96 | 40.62 | 25.00 | 35.42 | 56.25 |
|
| 17 |
+
| Scoped evidence | 96 | **96.88** | **78.12** | 65.62 | 47.92 | 55.21 | 47.92 | 51.04 | **79.17** | 25.00 | 10.42 | 36.46 | 25.00 | 37.50 | 36.46 | 89.58 |
|
| 18 |
+
| State tracking | 96 | 28.12 | 35.42 | 38.54 | 34.38 | 31.25 | 29.17 | 28.12 | 27.08 | 38.54 | 31.25 | 25.00 | 29.17 | 23.96 | 29.17 | 33.33 |
|
| 19 |
| In / out of menu | 64 | **100.00** | **100.00** | 84.38 | 75.00 | 71.88 | 59.38 | 60.94 | **100.00** | 70.31 | 57.81 | 62.50 | 75.00 | 64.06 | 37.50 | 100.00 |
|
| 20 |
|
| 21 |
</details>
|
|
|
|
| 26 |
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 27 |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 28 |
| Record identity | 80 | **61.25** | 53.75 | 45.00 | 50.00 | 57.50 | 56.25 | 50.00 | 50.00 | **60.00** | 50.00 | 46.25 | 51.25 | 47.50 | 50.00 | 76.25 |
|
| 29 |
+
| Capacity assignment | 80 | **73.75** | 46.25 | 50.00 | 50.00 | 50.00 | 51.25 | 50.00 | 47.50 | 50.00 | 50.00 | 50.00 | 50.00 | 63.75 | 50.00 | 68.75 |
|
| 30 |
| Constraint assignment | 80 | 32.50 | **41.25** | 27.50 | 32.50 | 25.00 | 23.75 | 23.75 | 26.25 | 25.00 | 20.00 | 21.25 | 26.25 | 22.50 | 27.50 | 55.00 |
|
| 31 |
| Intent routing · EN | 120 | 90.00 | 91.67 | 86.67 | 87.50 | 88.33 | 90.00 | 91.67 | 90.83 | 91.67 | 83.33 | 70.83 | 81.67 | 74.17 | 70.83 | 91.67 |
|
| 32 |
| Intent routing · ZH | 120 | 87.50 | 87.50 | 85.83 | 86.67 | 86.67 | 88.33 | 86.67 | 87.50 | 85.00 | 85.83 | 74.17 | 79.17 | 44.17 | 73.33 | 88.33 |
|
| 33 |
| Multiset reconciliation | 80 | 28.75 | 28.75 | 37.50 | 28.75 | 20.00 | 23.75 | 27.50 | 28.75 | 26.25 | 25.00 | 27.50 | 25.00 | 26.25 | 32.50 | 58.75 |
|
| 34 |
+
| Ordered service loss | 80 | 26.25 | **30.00** | 25.00 | 20.00 | 20.00 | 22.50 | 21.25 | 25.00 | 20.00 | 27.50 | 20.00 | 20.00 | 20.00 | 17.50 | 42.50 |
|
| 35 |
| Conflicting rule closure | 80 | **36.25** | 33.75 | 27.50 | 33.75 | 27.50 | 26.25 | 25.00 | 25.00 | 27.50 | 27.50 | 27.50 | 25.00 | 18.75 | 20.00 | 66.25 |
|
| 36 |
+
| Temporal exclusion | 80 | 37.50 | 42.50 | 28.75 | 45.00 | 36.25 | 42.50 | 31.25 | 41.25 | 32.50 | 21.25 | 28.75 | 17.50 | 12.50 | 28.75 | 41.25 |
|
| 37 |
+
| Transaction recovery | 80 | **53.75** | **62.50** | 43.75 | 51.25 | 35.00 | 41.25 | 26.25 | 38.75 | 42.50 | 32.50 | 23.75 | 32.50 | 23.75 | 18.75 | 75.00 |
|
| 38 |
|
| 39 |
</details>
|
| 40 |
|
|
|
|
| 43 |
|
| 44 |
| Task | n | Lux-9B | Nox-4B | Kev-9B | Kev-4B | Qwen3.5-9B | Decider | Qwen3.5-4B | Sol-2B | Eos-0.8B | Kev-0.8B | Qwen3.5-2B | Kai-0.6B | Laya · English | Laya · Multilingual | Jev |
|
| 45 |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
|
| 46 |
+
| Yes / no reading | 160 | 89.38 | 86.25 | 91.25 | 88.75 | 84.38 | 91.88 | 83.75 | 83.12 | 75.62 | 74.38 | 65.00 | 74.38 | 69.38 | 69.38 | 92.50 |
|
| 47 |
| Reading · EN | 160 | 92.50 | 73.12 | 83.12 | 75.62 | 98.75 | 93.12 | 93.12 | 70.00 | 68.12 | 63.75 | 83.75 | 38.12 | 38.12 | 36.25 | 96.88 |
|
| 48 |
+
| Reading · ZH | 160 | 89.38 | 70.62 | 81.25 | 74.38 | 91.88 | 91.25 | 91.25 | 70.00 | 61.88 | 58.75 | 81.25 | 31.87 | 28.75 | 28.12 | 96.25 |
|
| 49 |
|
| 50 |
</details>
|
| 51 |
|
|
|
|
| 57 |
| Contextual reasoning | 120 | **82.50** | 70.83 | 68.33 | 70.83 | 64.17 | 66.67 | 60.83 | 69.17 | 70.00 | 42.50 | 56.67 | 46.67 | 30.00 | 20.00 | 86.67 |
|
| 58 |
| Answerability | 120 | **93.33** | **88.33** | 78.33 | 81.67 | 74.17 | 84.17 | 81.67 | 83.33 | 81.67 | 66.67 | 82.50 | 55.83 | 57.50 | 55.00 | 90.83 |
|
| 59 |
| Textual entailment | 120 | **92.50** | 89.17 | 90.83 | 89.17 | 81.67 | 90.83 | 80.00 | 88.33 | 84.17 | 77.50 | 64.17 | 78.33 | 72.50 | 72.50 | 82.50 |
|
| 60 |
+
| Scientific inference | 120 | 97.50 | 96.67 | 96.67 | 96.67 | 98.33 | 95.83 | 96.67 | 95.83 | 90.83 | 88.33 | 85.83 | 98.33 | 95.00 | 81.67 | 99.17 |
|
| 61 |
|
| 62 |
</details>
|
| 63 |
|
|
|
|
| 75 |
| Policy negation | 32 | **96.88** | 87.50 | 90.62 | 87.50 | 62.50 | 62.50 | 53.12 | 71.88 | 68.75 | 71.88 | 46.88 | 56.25 | 53.12 | 50.00 | 90.62 |
|
| 76 |
| Authorization contrast | 40 | 100.00 | 97.50 | 100.00 | 100.00 | 100.00 | 100.00 | 97.50 | 50.00 | 50.00 | 97.50 | 62.50 | 50.00 | 67.50 | 50.00 | 100.00 |
|
| 77 |
| Deadline contrast | 40 | **90.00** | 70.00 | 87.50 | 75.00 | 77.50 | 30.00 | 52.50 | 40.00 | 50.00 | 45.00 | 25.00 | 50.00 | 35.00 | 22.50 | 92.50 |
|
| 78 |
+
| Emotion | 80 | 60.00 | 60.00 | 67.50 | 65.00 | 52.50 | 86.25 | 65.00 | 22.50 | 45.00 | 60.00 | 67.50 | 53.75 | 62.50 | 56.25 | 67.50 |
|
| 79 |
+
| MMLU | 80 | 75.00 | 61.25 | 75.00 | 68.75 | 73.75 | 62.50 | 66.25 | 52.50 | 47.50 | 51.25 | 53.75 | 37.50 | 22.50 | 27.50 | 88.75 |
|
| 80 |
| MMLU-Pro | 200 | 53.00 | 37.00 | 53.00 | 45.50 | 53.50 | 37.50 | 45.00 | 24.50 | 17.00 | 22.50 | 28.00 | 12.50 | 11.00 | 11.50 | 84.00 |
|
| 81 |
+
| Paraphrase | 80 | **91.25** | 85.00 | 81.25 | 81.25 | 88.75 | 77.50 | 83.75 | 80.00 | 71.25 | 58.75 | 70.00 | 47.50 | 86.25 | 75.00 | 87.50 |
|
| 82 |
| Question entailment | 80 | 91.25 | 91.25 | 95.00 | 91.25 | 90.00 | 87.50 | 87.50 | 83.75 | 81.25 | 81.25 | 63.75 | 71.25 | 80.00 | 73.75 | 91.25 |
|
| 83 |
| Science questions | 80 | 100.00 | 98.75 | 100.00 | 100.00 | 100.00 | 98.75 | 98.75 | 97.50 | 97.50 | 96.25 | 96.25 | 90.00 | 90.00 | 72.50 | 100.00 |
|
| 84 |
| Offensive-language detection | 80 | 75.00 | 75.00 | 86.25 | 85.00 | 83.75 | 88.75 | 77.50 | 75.00 | 52.50 | 72.50 | 83.75 | 78.75 | 81.25 | 82.50 | 76.25 |
|
WEIGHTING.md
CHANGED
|
@@ -4,7 +4,7 @@ The current product-priority weights were chosen after observing results. This c
|
|
| 4 |
|
| 5 |
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
-
| Lux-9B | 77.
|
| 8 |
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
|
|
|
| 4 |
|
| 5 |
| Model | Current 30/25/15/15/15 | Prior 25/25/15/15/20 | Original four-panel mean |
|
| 6 |
|---|---:|---:|---:|
|
| 7 |
+
| Lux-9B | 77.40 | 77.07 | 79.69 |
|
| 8 |
| Nox-4B | 73.09 | 72.42 | 75.03 |
|
| 9 |
| Kev-9B | 71.89 | 72.01 | 73.19 |
|
| 10 |
| Kev-4B | 70.09 | 70.30 | 71.73 |
|
assets/decision-matrix.pdf
CHANGED
|
Binary files a/assets/decision-matrix.pdf and b/assets/decision-matrix.pdf differ
|
|
|
assets/decision-matrix.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
assets/decision-matrix.svg
CHANGED
|
|
|
|
assets/decision-ranking.pdf
CHANGED
|
Binary files a/assets/decision-ranking.pdf and b/assets/decision-ranking.pdf differ
|
|
|
assets/decision-ranking.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
assets/decision-ranking.svg
CHANGED
|
|
|
|
metrics/benchmark.json
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
metrics/evaluation-provenance.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
|
| 3 |
-
"merged_statistics_sha256": "
|
| 4 |
"source_per_model": {
|
| 5 |
"Jev": {
|
| 6 |
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
|
@@ -111,9 +111,9 @@
|
|
| 111 |
}
|
| 112 |
},
|
| 113 |
"Lux": {
|
| 114 |
-
"statistics_sha256": "
|
| 115 |
-
"source_model_key": "Lux-
|
| 116 |
-
"manifest_sha256": "
|
| 117 |
"qualified_runtime": {
|
| 118 |
"python": "3.12.13",
|
| 119 |
"numpy": "2.3.5"
|
|
@@ -157,27 +157,27 @@
|
|
| 157 |
"shared_reference_rows_exact": [
|
| 158 |
[
|
| 159 |
"Nox",
|
| 160 |
-
"
|
| 161 |
],
|
| 162 |
[
|
| 163 |
"Nox",
|
| 164 |
-
"
|
| 165 |
],
|
| 166 |
[
|
| 167 |
"Nox",
|
| 168 |
-
"
|
| 169 |
],
|
| 170 |
[
|
| 171 |
"Nox",
|
| 172 |
-
"
|
| 173 |
],
|
| 174 |
[
|
| 175 |
"Nox",
|
| 176 |
-
"
|
| 177 |
],
|
| 178 |
[
|
| 179 |
"Nox",
|
| 180 |
-
"
|
| 181 |
],
|
| 182 |
[
|
| 183 |
"Nox",
|
|
@@ -185,59 +185,59 @@
|
|
| 185 |
],
|
| 186 |
[
|
| 187 |
"Nox",
|
| 188 |
-
"
|
| 189 |
],
|
| 190 |
[
|
| 191 |
"Nox",
|
| 192 |
-
"
|
| 193 |
],
|
| 194 |
[
|
| 195 |
"Nox",
|
| 196 |
-
"
|
| 197 |
],
|
| 198 |
[
|
| 199 |
"Nox",
|
| 200 |
-
"
|
| 201 |
],
|
| 202 |
[
|
| 203 |
"Nox",
|
| 204 |
-
"
|
| 205 |
],
|
| 206 |
[
|
| 207 |
"Nox",
|
| 208 |
-
"
|
| 209 |
],
|
| 210 |
[
|
| 211 |
"Nox",
|
| 212 |
-
"
|
| 213 |
],
|
| 214 |
[
|
| 215 |
"Nox",
|
| 216 |
-
"
|
| 217 |
],
|
| 218 |
[
|
| 219 |
"Lux",
|
| 220 |
-
"
|
| 221 |
],
|
| 222 |
[
|
| 223 |
"Lux",
|
| 224 |
-
"
|
| 225 |
],
|
| 226 |
[
|
| 227 |
"Lux",
|
| 228 |
-
"
|
| 229 |
],
|
| 230 |
[
|
| 231 |
"Lux",
|
| 232 |
-
"
|
| 233 |
],
|
| 234 |
[
|
| 235 |
"Lux",
|
| 236 |
-
"
|
| 237 |
],
|
| 238 |
[
|
| 239 |
"Lux",
|
| 240 |
-
"
|
| 241 |
],
|
| 242 |
[
|
| 243 |
"Lux",
|
|
@@ -245,67 +245,67 @@
|
|
| 245 |
],
|
| 246 |
[
|
| 247 |
"Lux",
|
| 248 |
-
"
|
| 249 |
],
|
| 250 |
[
|
| 251 |
"Lux",
|
| 252 |
-
"
|
| 253 |
],
|
| 254 |
[
|
| 255 |
"Lux",
|
| 256 |
-
"
|
| 257 |
],
|
| 258 |
[
|
| 259 |
"Lux",
|
| 260 |
-
"
|
| 261 |
],
|
| 262 |
[
|
| 263 |
"Lux",
|
| 264 |
-
"
|
| 265 |
],
|
| 266 |
[
|
| 267 |
"Lux",
|
| 268 |
-
"
|
| 269 |
],
|
| 270 |
[
|
| 271 |
"Lux",
|
| 272 |
-
"
|
| 273 |
],
|
| 274 |
[
|
| 275 |
"Lux",
|
| 276 |
-
"
|
| 277 |
],
|
| 278 |
[
|
| 279 |
"Kai",
|
| 280 |
-
"
|
| 281 |
],
|
| 282 |
[
|
| 283 |
"Kai",
|
| 284 |
-
"
|
| 285 |
],
|
| 286 |
[
|
| 287 |
"Kai",
|
| 288 |
-
"
|
| 289 |
],
|
| 290 |
[
|
| 291 |
"Kai",
|
| 292 |
-
"
|
| 293 |
],
|
| 294 |
[
|
| 295 |
"Eos",
|
| 296 |
-
"
|
| 297 |
],
|
| 298 |
[
|
| 299 |
"Eos",
|
| 300 |
-
"
|
| 301 |
],
|
| 302 |
[
|
| 303 |
"Eos",
|
| 304 |
-
"
|
| 305 |
],
|
| 306 |
[
|
| 307 |
"Eos",
|
| 308 |
-
"
|
| 309 |
]
|
| 310 |
],
|
| 311 |
"published_matrices_exact": {
|
|
@@ -319,5 +319,5 @@
|
|
| 319 |
}
|
| 320 |
}
|
| 321 |
},
|
| 322 |
-
"source_builder_sha256": "
|
| 323 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"frozen_protocol_sha256": "ab97f652ab890eaefdd5373dd2deeb1e26588815f1f339eee4bcaa610846b0af",
|
| 3 |
+
"merged_statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
|
| 4 |
"source_per_model": {
|
| 5 |
"Jev": {
|
| 6 |
"statistics_sha256": "b319a993ac569ea1fa8da027ccb2195868557e4efe232a5d06091db2884086be",
|
|
|
|
| 111 |
}
|
| 112 |
},
|
| 113 |
"Lux": {
|
| 114 |
+
"statistics_sha256": "71251a31ed6435d8c94df6d2a5241e9183613c19a87f8970145afec8ec948efc",
|
| 115 |
+
"source_model_key": "Lux-reading-source-candidate",
|
| 116 |
+
"manifest_sha256": "7107a2a8264a0416fa75fe40e57f3d06e4cb0395527eaedfa7b7335d8e37109a",
|
| 117 |
"qualified_runtime": {
|
| 118 |
"python": "3.12.13",
|
| 119 |
"numpy": "2.3.5"
|
|
|
|
| 157 |
"shared_reference_rows_exact": [
|
| 158 |
[
|
| 159 |
"Nox",
|
| 160 |
+
"llm2jev-2b"
|
| 161 |
],
|
| 162 |
[
|
| 163 |
"Nox",
|
| 164 |
+
"kev-9b"
|
| 165 |
],
|
| 166 |
[
|
| 167 |
"Nox",
|
| 168 |
+
"Sol"
|
| 169 |
],
|
| 170 |
[
|
| 171 |
"Nox",
|
| 172 |
+
"Nox-null-description-candidate"
|
| 173 |
],
|
| 174 |
[
|
| 175 |
"Nox",
|
| 176 |
+
"Qwen3.5-2B"
|
| 177 |
],
|
| 178 |
[
|
| 179 |
"Nox",
|
| 180 |
+
"Qwen3.5-9B"
|
| 181 |
],
|
| 182 |
[
|
| 183 |
"Nox",
|
|
|
|
| 185 |
],
|
| 186 |
[
|
| 187 |
"Nox",
|
| 188 |
+
"kev-4b"
|
| 189 |
],
|
| 190 |
[
|
| 191 |
"Nox",
|
| 192 |
+
"Decider"
|
| 193 |
],
|
| 194 |
[
|
| 195 |
"Nox",
|
| 196 |
+
"Laya-base"
|
| 197 |
],
|
| 198 |
[
|
| 199 |
"Nox",
|
| 200 |
+
"Jev"
|
| 201 |
],
|
| 202 |
[
|
| 203 |
"Nox",
|
| 204 |
+
"Laya-multilingual"
|
| 205 |
],
|
| 206 |
[
|
| 207 |
"Nox",
|
| 208 |
+
"llm2jev-4b"
|
| 209 |
],
|
| 210 |
[
|
| 211 |
"Nox",
|
| 212 |
+
"nimble-9b"
|
| 213 |
],
|
| 214 |
[
|
| 215 |
"Nox",
|
| 216 |
+
"Qwen3.5-4B"
|
| 217 |
],
|
| 218 |
[
|
| 219 |
"Lux",
|
| 220 |
+
"llm2jev-2b"
|
| 221 |
],
|
| 222 |
[
|
| 223 |
"Lux",
|
| 224 |
+
"kev-9b"
|
| 225 |
],
|
| 226 |
[
|
| 227 |
"Lux",
|
| 228 |
+
"Sol"
|
| 229 |
],
|
| 230 |
[
|
| 231 |
"Lux",
|
| 232 |
+
"Nox-null-description-candidate"
|
| 233 |
],
|
| 234 |
[
|
| 235 |
"Lux",
|
| 236 |
+
"Qwen3.5-2B"
|
| 237 |
],
|
| 238 |
[
|
| 239 |
"Lux",
|
| 240 |
+
"Qwen3.5-9B"
|
| 241 |
],
|
| 242 |
[
|
| 243 |
"Lux",
|
|
|
|
| 245 |
],
|
| 246 |
[
|
| 247 |
"Lux",
|
| 248 |
+
"kev-4b"
|
| 249 |
],
|
| 250 |
[
|
| 251 |
"Lux",
|
| 252 |
+
"Decider"
|
| 253 |
],
|
| 254 |
[
|
| 255 |
"Lux",
|
| 256 |
+
"Laya-base"
|
| 257 |
],
|
| 258 |
[
|
| 259 |
"Lux",
|
| 260 |
+
"Jev"
|
| 261 |
],
|
| 262 |
[
|
| 263 |
"Lux",
|
| 264 |
+
"Laya-multilingual"
|
| 265 |
],
|
| 266 |
[
|
| 267 |
"Lux",
|
| 268 |
+
"llm2jev-4b"
|
| 269 |
],
|
| 270 |
[
|
| 271 |
"Lux",
|
| 272 |
+
"nimble-9b"
|
| 273 |
],
|
| 274 |
[
|
| 275 |
"Lux",
|
| 276 |
+
"Qwen3.5-4B"
|
| 277 |
],
|
| 278 |
[
|
| 279 |
"Kai",
|
| 280 |
+
"Laya-base"
|
| 281 |
],
|
| 282 |
[
|
| 283 |
"Kai",
|
| 284 |
+
"Laya-multilingual"
|
| 285 |
],
|
| 286 |
[
|
| 287 |
"Kai",
|
| 288 |
+
"kev-0.8b"
|
| 289 |
],
|
| 290 |
[
|
| 291 |
"Kai",
|
| 292 |
+
"Qwen3.5-2B"
|
| 293 |
],
|
| 294 |
[
|
| 295 |
"Eos",
|
| 296 |
+
"Laya-base"
|
| 297 |
],
|
| 298 |
[
|
| 299 |
"Eos",
|
| 300 |
+
"Laya-multilingual"
|
| 301 |
],
|
| 302 |
[
|
| 303 |
"Eos",
|
| 304 |
+
"kev-0.8b"
|
| 305 |
],
|
| 306 |
[
|
| 307 |
"Eos",
|
| 308 |
+
"Qwen3.5-2B"
|
| 309 |
]
|
| 310 |
],
|
| 311 |
"published_matrices_exact": {
|
|
|
|
| 319 |
}
|
| 320 |
}
|
| 321 |
},
|
| 322 |
+
"source_builder_sha256": "6d7aab8b701df4fe67750a2e0212fcc1e6321b4dad71d5723b3a2d6fd97e1ecd"
|
| 323 |
}
|
release-manifest.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"status": "qualified-runtime-and-current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 5 |
"readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
|
| 6 |
-
"model_card_sha256": "
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
|
@@ -22,7 +22,7 @@
|
|
| 22 |
{
|
| 23 |
"file": "DIAGNOSTICS.md",
|
| 24 |
"bytes": 6327,
|
| 25 |
-
"sha256": "
|
| 26 |
},
|
| 27 |
{
|
| 28 |
"file": "Dockerfile.runtime",
|
|
@@ -32,7 +32,7 @@
|
|
| 32 |
{
|
| 33 |
"file": "EVALUATION.md",
|
| 34 |
"bytes": 4079,
|
| 35 |
-
"sha256": "
|
| 36 |
},
|
| 37 |
{
|
| 38 |
"file": "LICENSE",
|
|
@@ -42,7 +42,7 @@
|
|
| 42 |
{
|
| 43 |
"file": "MATERIALS.json",
|
| 44 |
"bytes": 2064,
|
| 45 |
-
"sha256": "
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
@@ -67,7 +67,7 @@
|
|
| 67 |
{
|
| 68 |
"file": "README.md",
|
| 69 |
"bytes": 6169,
|
| 70 |
-
"sha256": "
|
| 71 |
},
|
| 72 |
{
|
| 73 |
"file": "RUNTIME-RELEASE.json",
|
|
@@ -87,7 +87,7 @@
|
|
| 87 |
{
|
| 88 |
"file": "SENSITIVITY.md",
|
| 89 |
"bytes": 866,
|
| 90 |
-
"sha256": "
|
| 91 |
},
|
| 92 |
{
|
| 93 |
"file": "SERVING_OPTIMIZATION.json",
|
|
@@ -102,7 +102,7 @@
|
|
| 102 |
{
|
| 103 |
"file": "TASKS.md",
|
| 104 |
"bytes": 10393,
|
| 105 |
-
"sha256": "
|
| 106 |
},
|
| 107 |
{
|
| 108 |
"file": "USAGE.md",
|
|
@@ -112,7 +112,7 @@
|
|
| 112 |
{
|
| 113 |
"file": "WEIGHTING.md",
|
| 114 |
"bytes": 866,
|
| 115 |
-
"sha256": "
|
| 116 |
},
|
| 117 |
{
|
| 118 |
"file": "assets/architecture.png",
|
|
@@ -251,18 +251,18 @@
|
|
| 251 |
},
|
| 252 |
{
|
| 253 |
"file": "assets/decision-matrix.pdf",
|
| 254 |
-
"bytes":
|
| 255 |
-
"sha256": "
|
| 256 |
},
|
| 257 |
{
|
| 258 |
"file": "assets/decision-matrix.png",
|
| 259 |
-
"bytes":
|
| 260 |
-
"sha256": "
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"file": "assets/decision-matrix.svg",
|
| 264 |
"bytes": 50107,
|
| 265 |
-
"sha256": "
|
| 266 |
},
|
| 267 |
{
|
| 268 |
"file": "assets/decision-nox-4b-header.png",
|
|
@@ -292,17 +292,17 @@
|
|
| 292 |
{
|
| 293 |
"file": "assets/decision-ranking.pdf",
|
| 294 |
"bytes": 25376,
|
| 295 |
-
"sha256": "
|
| 296 |
},
|
| 297 |
{
|
| 298 |
"file": "assets/decision-ranking.png",
|
| 299 |
-
"bytes":
|
| 300 |
-
"sha256": "
|
| 301 |
},
|
| 302 |
{
|
| 303 |
"file": "assets/decision-ranking.svg",
|
| 304 |
"bytes": 15728,
|
| 305 |
-
"sha256": "
|
| 306 |
},
|
| 307 |
{
|
| 308 |
"file": "assets/readout.png",
|
|
@@ -381,8 +381,8 @@
|
|
| 381 |
},
|
| 382 |
{
|
| 383 |
"file": "metrics/benchmark.json",
|
| 384 |
-
"bytes":
|
| 385 |
-
"sha256": "
|
| 386 |
},
|
| 387 |
{
|
| 388 |
"file": "metrics/comparator-coverage.json",
|
|
@@ -391,8 +391,8 @@
|
|
| 391 |
},
|
| 392 |
{
|
| 393 |
"file": "metrics/evaluation-provenance.json",
|
| 394 |
-
"bytes":
|
| 395 |
-
"sha256": "
|
| 396 |
},
|
| 397 |
{
|
| 398 |
"file": "metrics/expanded-quality.json",
|
|
@@ -580,9 +580,9 @@
|
|
| 580 |
"scope": "Latest qualified Lux comparison and54 task diagnostics; all other model rows and inference files unchanged.",
|
| 581 |
"release_tag": "v1.3.2",
|
| 582 |
"change_kind": "latest-comparison-documents-only",
|
| 583 |
-
"previous_main_revision": "
|
| 584 |
-
"previous_release_manifest_sha256": "
|
| 585 |
-
"presentation_amendment_sha256": "
|
| 586 |
-
"statistics_sha256": "
|
| 587 |
"latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
|
| 588 |
}
|
|
|
|
| 3 |
"status": "qualified-runtime-and-current-documents-assembled",
|
| 4 |
"bundle_manifest_sha256": "7d9b06bc25a75b1f131aabd280df0ef2777bf69db98574a300067ace40c61c9c",
|
| 5 |
"readiness_sha256": "fcbb84a9bf7294ac79e2421cdd2cbfdfc46a79fd74c41ba0965778472d1cafc1",
|
| 6 |
+
"model_card_sha256": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5",
|
| 7 |
"repo_id": "llm-semantic-router/Decision-1.0-Nox-4B",
|
| 8 |
"assembly_script_sha256": "a8f72b8a67a64034d6183f89c945bab611056816ba0cba43670e9becb6eb78c4",
|
| 9 |
"original_bundle_manifest_preserved": false,
|
|
|
|
| 22 |
{
|
| 23 |
"file": "DIAGNOSTICS.md",
|
| 24 |
"bytes": 6327,
|
| 25 |
+
"sha256": "cc8c0a378013d1cfb87ec775344180047136109cbc236a23da02201c0a7c5baa"
|
| 26 |
},
|
| 27 |
{
|
| 28 |
"file": "Dockerfile.runtime",
|
|
|
|
| 32 |
{
|
| 33 |
"file": "EVALUATION.md",
|
| 34 |
"bytes": 4079,
|
| 35 |
+
"sha256": "10744aeed1a113597e644d0b3f4787a91299f055701f22c8cd34843276a7df2a"
|
| 36 |
},
|
| 37 |
{
|
| 38 |
"file": "LICENSE",
|
|
|
|
| 42 |
{
|
| 43 |
"file": "MATERIALS.json",
|
| 44 |
"bytes": 2064,
|
| 45 |
+
"sha256": "483718bbc312bfc1311db9900d7767d5a8bb9ac69fb87edf89a95d4d15b80eea"
|
| 46 |
},
|
| 47 |
{
|
| 48 |
"file": "NORMALIZATION_RUNTIME.md",
|
|
|
|
| 67 |
{
|
| 68 |
"file": "README.md",
|
| 69 |
"bytes": 6169,
|
| 70 |
+
"sha256": "ff11f99561dea4a557bf8b5f135374bfbd79c576beaf9f74d92aa94971d6ada5"
|
| 71 |
},
|
| 72 |
{
|
| 73 |
"file": "RUNTIME-RELEASE.json",
|
|
|
|
| 87 |
{
|
| 88 |
"file": "SENSITIVITY.md",
|
| 89 |
"bytes": 866,
|
| 90 |
+
"sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5"
|
| 91 |
},
|
| 92 |
{
|
| 93 |
"file": "SERVING_OPTIMIZATION.json",
|
|
|
|
| 102 |
{
|
| 103 |
"file": "TASKS.md",
|
| 104 |
"bytes": 10393,
|
| 105 |
+
"sha256": "573086057e6d87135f119ad95fda16a46f9e8a5e1109430f0a1b8a5da7d99247"
|
| 106 |
},
|
| 107 |
{
|
| 108 |
"file": "USAGE.md",
|
|
|
|
| 112 |
{
|
| 113 |
"file": "WEIGHTING.md",
|
| 114 |
"bytes": 866,
|
| 115 |
+
"sha256": "6bda57a973be2c1afcca5f579b37d1979fa89ff0b1ef171b6d8414e6215d2bc5"
|
| 116 |
},
|
| 117 |
{
|
| 118 |
"file": "assets/architecture.png",
|
|
|
|
| 251 |
},
|
| 252 |
{
|
| 253 |
"file": "assets/decision-matrix.pdf",
|
| 254 |
+
"bytes": 29410,
|
| 255 |
+
"sha256": "3b4f10c3f00820d4b82b11a48401a8c360304c4e7cda90a6e7133c76d18ca41b"
|
| 256 |
},
|
| 257 |
{
|
| 258 |
"file": "assets/decision-matrix.png",
|
| 259 |
+
"bytes": 375860,
|
| 260 |
+
"sha256": "cb0c583975fc6848f0d3ca7ed11da48347f7431f91f4debf699b3fb039a0a696"
|
| 261 |
},
|
| 262 |
{
|
| 263 |
"file": "assets/decision-matrix.svg",
|
| 264 |
"bytes": 50107,
|
| 265 |
+
"sha256": "fc83d1c36264098fce20ca9f000aedc04473c7a16295fa5d16a82afeee2342fc"
|
| 266 |
},
|
| 267 |
{
|
| 268 |
"file": "assets/decision-nox-4b-header.png",
|
|
|
|
| 292 |
{
|
| 293 |
"file": "assets/decision-ranking.pdf",
|
| 294 |
"bytes": 25376,
|
| 295 |
+
"sha256": "e2d6ecdab0599e65db77b520669e90b32fad6d10c85bb15dcd77bec1ed6ec51f"
|
| 296 |
},
|
| 297 |
{
|
| 298 |
"file": "assets/decision-ranking.png",
|
| 299 |
+
"bytes": 248285,
|
| 300 |
+
"sha256": "9bb5e1487f2b66215624e67a72bda0d1c3f4ad6cd931089aa5bd7211f3be02b9"
|
| 301 |
},
|
| 302 |
{
|
| 303 |
"file": "assets/decision-ranking.svg",
|
| 304 |
"bytes": 15728,
|
| 305 |
+
"sha256": "c8600115c3fafbf03e0b06cc6e3da016e4f74473df4f2d4575ae1c9721c99b8e"
|
| 306 |
},
|
| 307 |
{
|
| 308 |
"file": "assets/readout.png",
|
|
|
|
| 381 |
},
|
| 382 |
{
|
| 383 |
"file": "metrics/benchmark.json",
|
| 384 |
+
"bytes": 2569391,
|
| 385 |
+
"sha256": "86dc156356ee80eb849808f6e0d5237157f63d121ae6992c9a833691912ef76c"
|
| 386 |
},
|
| 387 |
{
|
| 388 |
"file": "metrics/comparator-coverage.json",
|
|
|
|
| 391 |
},
|
| 392 |
{
|
| 393 |
"file": "metrics/evaluation-provenance.json",
|
| 394 |
+
"bytes": 8510,
|
| 395 |
+
"sha256": "f9c087bd92e6cdfdc4c4dabc0ec335bb837a79715577a7355012d3a21f81c8e9"
|
| 396 |
},
|
| 397 |
{
|
| 398 |
"file": "metrics/expanded-quality.json",
|
|
|
|
| 580 |
"scope": "Latest qualified Lux comparison and54 task diagnostics; all other model rows and inference files unchanged.",
|
| 581 |
"release_tag": "v1.3.2",
|
| 582 |
"change_kind": "latest-comparison-documents-only",
|
| 583 |
+
"previous_main_revision": "bb411db257ca34a99a085764fcf44d3a7c7c2874",
|
| 584 |
+
"previous_release_manifest_sha256": "8063ae6e1048516567577001f922d1ed6abf8701f5fc4888166330c9bedbaa40",
|
| 585 |
+
"presentation_amendment_sha256": "4fadc67b276d2b1e79e19ccacd383482cb0580996450d0103ffe2a90cb494a56",
|
| 586 |
+
"statistics_sha256": "8406aea215dc1c4ee2645130c94472b336bb065ba46c6efac4f742589499463b",
|
| 587 |
"latency_correction_sha256": "9727f7a0768d3888ed8cd22f95b25598e558b447adc8f9f4f5a0bb24959269b1"
|
| 588 |
}
|