Synchronizing local compiler cache.
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +149 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/43f30d107729db6b760b2a90833f43ca44f68e6468ef64e131f71e2fa9f5b23f/b76d86cc7c31e0e29c99.json +104 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/43f30d107729db6b760b2a90833f43ca44f68e6468ef64e131f71e2fa9f5b23f/f71907dd62c8b40e471f.json +104 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/gpt_oss/openai/gpt-oss-20b/b76d86cc7c31e0e29c99.json +104 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/0cbac419f8839d90c6bafb19e17441f1a052e93e227f0fa62918ebe7d882e225/88fa555c8659e3f7fc5a.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/009b59fa3cc87705bbb4.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/06dea0fe4ee55a8035aa.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/0ccfeb749c49d4002400.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/0d6dd8f35029b2597ea3.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/12054fb3f8fa1b1fc9a6.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/22e7cd94e28b353a796b.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/26b9c01e8f46a57f8933.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/2f14fea94ed53b3b1d94.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/34bc113dd21cb74d6179.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/44994da16f213a28fa4c.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/49052c1d22e88f219887.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/49fc8d2feaf8eca91a07.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/535e7795fb656220e11a.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/55d254e5fde0dcab0e41.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/5e301b3a72a832a33468.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/644bf9550db2b89192aa.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/6501938b047f7d373cdd.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/67070f0a4d500338d5aa.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/7bfaacc5ae3961c38f6c.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/824782edf021538e230f.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/8c7328c05cd751a24e18.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/8fdc4765f723aa1ce54e.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/912ca3353189929b7b15.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/99d0c0f90ad212bff9e2.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/99d672cbaa2e018da4e4.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/9d23ffa4ccbadb0a623e.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/9e94643f5e1e669914b1.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a41e0220fcb4438b8502.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a45b5270d576ce1c1a2a.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a52b33ce4a3bb2a7e4bb.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a8ca7ace639199dfc385.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/ad354086b250f133c9c6.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/aec994126b22dffefd0b.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b0bfc6ba654a35354148.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b2f13b35b2326e133272.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b4a6b1d49ffe1fb09037.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b52dd6b442b63b75e7b7.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b8a28a3ac7bcba98b595.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/bba41446ac8406c873ee.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/c2f89131f4ebe4bff600.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/c67e5e7f18f7dbf76e0c.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/cade97aae05512df69b7.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d0e85bdeabc9387b9465.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d27d49077cbb0bf50eb9.json +95 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d59c183509c2a322fd8b.json +95 -0
.gitattributes
CHANGED
|
@@ -6581,3 +6581,152 @@ neuronxcc-2.21.33363.0+82129205/MODULE_ce7e2a2399f408878a1f+24129607/model.neff
|
|
| 6581 |
neuronxcc-2.21.33363.0+82129205/MODULE_f9b5b25ceb35d0bb7f65+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6582 |
neuronxcc-2.21.33363.0+82129205/MODULE_f9b5b25ceb35d0bb7f65+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6583 |
neuronxcc-2.21.33363.0+82129205/MODULE_3f31618ecae0155b9051+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6581 |
neuronxcc-2.21.33363.0+82129205/MODULE_f9b5b25ceb35d0bb7f65+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6582 |
neuronxcc-2.21.33363.0+82129205/MODULE_f9b5b25ceb35d0bb7f65+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6583 |
neuronxcc-2.21.33363.0+82129205/MODULE_3f31618ecae0155b9051+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6584 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_00246f51d08f7ed27839+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6585 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_00fa01e7f21f195c4444+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6586 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_02007c49982251cdfd74+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6587 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0b5ceaefeeecbe146efd+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6588 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0bf0f7b24eb5eb88ff95+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6589 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0cbceb2b35b400eada58+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6590 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0cef33f7c9fcedab6d7c+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6591 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0cef33f7c9fcedab6d7c+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6592 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0d689ac638673eadd329+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6593 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0f17ee81390d681c2228+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6594 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0fed968161ee53fb93ea+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6595 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_11c8f427ce1ae4c2006e+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6596 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_12461911462967760525+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6597 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_12a28df006c59e3576b1+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6598 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_12a28df006c59e3576b1+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6599 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_14192315105465839210+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6600 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1582447fecd752c400bc+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6601 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1605570184959188488+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6602 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_17516957483816756456+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6603 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_17567634454620665439+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6604 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1ab71ba131418d320f9b+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6605 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1c279d75c64f4c7f28cb+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6606 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1d3dabdabdfa7a5c0ca9+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6607 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1fb1d66f34fd07cc4310+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6608 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_210b72f81fca8bd7952f+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6609 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_22c326cc7ae69de46c3c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6610 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_249689642401186199+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6611 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2749861b5474cc5ff87c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6612 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2831e4c199ca2002f484+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6613 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_28ddfe81fe61b9dbddf0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6614 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_29374f693146a8400712+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6615 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_294c539670d18fca3b7a+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6616 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2ad6bb08c01907e38254+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6617 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2bbf2248c52ac915eb6f+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6618 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2c8b27eef6e9052806fe+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6619 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2de7ac18660b7ecd4dcb+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6620 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3034f0c01273e0fddb60+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6621 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_308568a6877c754a5693+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6622 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_329d10715e3353b4c19e+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6623 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_34119e2d07794a4f0701+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6624 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_35958e2f8cf19e5eddc6+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6625 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_389e4ac7e4281a32a38a+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6626 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3a22f3795f95b4bc7eef+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6627 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3a3d29278a0f2a177084+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6628 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4310522466763f19e4a9+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6629 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_446d3822c1ba9fbe18ba+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6630 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_44a91259c06aa0b69cdb+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6631 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4e90b6140446b35617df+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6632 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4e9f100f0dcd09bff402+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6633 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4f6bc8a0a1a483ba7b55+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6634 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_52d550ce85cfe5f7cde5+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6635 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_52e7224a0c22a92654eb+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6636 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_532d27607bd8cdc47997+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6637 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_535b06a6f27a7161adcf+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6638 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_541e319c815dc74f5f11+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6639 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5673c277e4f5fe954d60+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6640 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_576ba585336d7db50d6d+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6641 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_57d25abe3c5f66cf1156+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6642 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_593ee0ced8d35810d75f+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6643 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5d24f683ef1fde5a5652+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6644 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5dafb05b2fa15b606ad4+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6645 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5dafb05b2fa15b606ad4+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6646 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_62a5b0034aeb1bd8056d+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6647 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_639d3c8ddff2950b55fe+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6648 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_64886f8b7709113eb14b+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6649 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_65b0c122726250bd8b02+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6650 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_66af75d88691306d5c36+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6651 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_691f40d8d6238f18e019+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6652 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6f655c6dbddaa46f86cb+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6653 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6f7d67366590b6d1215b+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6654 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6f7d67366590b6d1215b+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6655 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_72772fa709f43f40424c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6656 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_745d089e75e795ad6a8b+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6657 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_758311bdb6e777cb53a2+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6658 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_75b1bf9bd90504d384d1+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6659 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_75cb0c08a4502feae194+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6660 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_76054337ad9a319bf520+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6661 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_78f3eeb078cd51748fc6+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6662 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_790812dbd057158980ea+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6663 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_7b11eb682e93341f0887+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6664 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_7eba564503196965e26c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6665 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_813f4ca5979c12f89e82+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6666 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_815bc4fb35f01af9f1ae+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6667 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_81f3ec270dd5aa6f295b+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6668 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_86b0f00722f218bdd733+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6669 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_86be7015514811130001+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6670 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_882cde96bdd79751820a+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6671 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_889e53ae9bd551bec0ad+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6672 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8ac6eb993143ee441f61+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6673 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8b9955761e61e2616e95+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6674 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8c2c4d30165b7df0ff60+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6675 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8cdf0668abc5ec3574a0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6676 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8dd69135fcd689bad9f9+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6677 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_906394eb7f18a4a73151+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6678 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_90b55091a26739499584+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6679 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_93667ce6adf087dd27b0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6680 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9590cb0e237c0f2b7303+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6681 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9644019637695497943+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6682 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9847227195640834860+e30acd3a/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6683 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_992f9d164a8d61e4a43a+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6684 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_99e9377627d0673637ab+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6685 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9b3270c73bfb15154d87+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6686 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a024648ac027e9e2c579+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6687 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a402787b8478f1fc5d77+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6688 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a45b47bc75d67a3051f8+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6689 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a4c75e33226b582ed13b+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6690 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a7f908df5e06d25c8068+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6691 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_aace4e3c5cf3d468cf2e+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6692 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_aaefb5ede5ee0772566c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6693 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_abf57419dc4fa1e355bf+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6694 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ada63286020c60a0372f+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6695 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ae3a7155c9cf865b3bf3+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6696 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_afebefb7fd2d147fcab6+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6697 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b0cd0a9c4d91cfdefe87+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6698 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b3633511af08e66ed5e5+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6699 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b3ac1cf6845fcd156f52+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6700 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b3ccc1c1e6b02608fbdd+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6701 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b3fedb970e36dce47927+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6702 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b7dbf963e5b5df662cbb+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6703 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_bd2f2fd229df31e1d6fc+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6704 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c1d12458349b6c6d46c0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6705 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c3d7a31c5ebb0fbecd0a+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6706 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c72c64369218eb78825d+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6707 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_cb4521984ab2c0a81467+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6708 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ccf17d106921a01f8900+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6709 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d4a7c9ec6145376f3d20+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6710 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d4b74a4b48943996d251+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6711 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d4b74a4b48943996d251+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 6712 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d9b4078117ed73d95bdc+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6713 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dc6e27c8b152ac1dc9ac+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6714 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dcd8c1453b43eed00c33+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6715 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_e275893e058d2cf4cbe5+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6716 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_e535e65447dbc9579a04+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6717 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_e615497c40e4471f01e2+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6718 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_edac86b9ffb002bc0f49+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6719 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_eeccb8e8c987751ce9b0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6720 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f081706679e4afd8547d+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6721 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f42fa724ddb12e6e9d97+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6722 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f511d700e615d9f77299+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6723 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f61d8c6ee177e8e759f0+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6724 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f699344365cf7f52a68f+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6725 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f81dd9a0c4854a438c30+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6726 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f90557dacb3ca5598321+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6727 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f93050c233c49c1a2125+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6728 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f9e37ca44ee2fbb4987c+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6729 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_fa42cc49802250e87f12+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6730 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_fc8e09ebb5e540fef7a3+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6731 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_fd0e003c65277ac495d4+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 6732 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_fdd8f98956e8286e2bbd+fb4cc044/model.neff filter=lfs diff=lfs merge=lfs -text
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/43f30d107729db6b760b2a90833f43ca44f68e6468ef64e131f71e2fa9f5b23f/b76d86cc7c31e0e29c99.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "openai/gpt-oss-20b",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GptOssForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": true,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"experts_per_token": 4,
|
| 12 |
+
"head_dim": 64,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2880,
|
| 15 |
+
"initial_context_length": 4096,
|
| 16 |
+
"initializer_range": 0.02,
|
| 17 |
+
"intermediate_size": 2880,
|
| 18 |
+
"layer_types": [
|
| 19 |
+
"sliding_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"sliding_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"sliding_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"sliding_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"sliding_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"sliding_attention",
|
| 42 |
+
"full_attention"
|
| 43 |
+
],
|
| 44 |
+
"max_position_embeddings": 131072,
|
| 45 |
+
"model_type": "gpt_oss",
|
| 46 |
+
"neuron": {
|
| 47 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 48 |
+
"batch_size": 1,
|
| 49 |
+
"capacity_factor": null,
|
| 50 |
+
"checkpoint_id": "openai/gpt-oss-20b",
|
| 51 |
+
"checkpoint_revision": "6cee5e81ee83917806bbde320786a8fb61efebee",
|
| 52 |
+
"continuous_batching": false,
|
| 53 |
+
"ep_degree": 1,
|
| 54 |
+
"fused_qkv": true,
|
| 55 |
+
"glu_mlp": true,
|
| 56 |
+
"local_ranks_size": 8,
|
| 57 |
+
"max_batch_size": 1,
|
| 58 |
+
"max_context_length": 4096,
|
| 59 |
+
"max_topk": 256,
|
| 60 |
+
"n_active_tokens": 4096,
|
| 61 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 62 |
+
"on_device_sampling": true,
|
| 63 |
+
"optimum_neuron_version": "0.4.5.dev1",
|
| 64 |
+
"output_logits": false,
|
| 65 |
+
"pp_degree": 1,
|
| 66 |
+
"sequence_length": 4096,
|
| 67 |
+
"speculation_length": 0,
|
| 68 |
+
"start_rank_id": 0,
|
| 69 |
+
"target": "trn1",
|
| 70 |
+
"torch_dtype": "bfloat16",
|
| 71 |
+
"tp_degree": 8
|
| 72 |
+
},
|
| 73 |
+
"num_attention_heads": 64,
|
| 74 |
+
"num_experts_per_tok": 4,
|
| 75 |
+
"num_hidden_layers": 24,
|
| 76 |
+
"num_key_value_heads": 8,
|
| 77 |
+
"num_local_experts": 32,
|
| 78 |
+
"output_router_logits": false,
|
| 79 |
+
"quantization_config": {
|
| 80 |
+
"modules_to_not_convert": [
|
| 81 |
+
"model.layers.*.self_attn",
|
| 82 |
+
"model.layers.*.mlp.router",
|
| 83 |
+
"model.embed_tokens",
|
| 84 |
+
"lm_head"
|
| 85 |
+
],
|
| 86 |
+
"quant_method": "mxfp4"
|
| 87 |
+
},
|
| 88 |
+
"rms_norm_eps": 1e-05,
|
| 89 |
+
"rope_scaling": {
|
| 90 |
+
"beta_fast": 32.0,
|
| 91 |
+
"beta_slow": 1.0,
|
| 92 |
+
"factor": 32.0,
|
| 93 |
+
"original_max_position_embeddings": 4096,
|
| 94 |
+
"rope_type": "yarn",
|
| 95 |
+
"truncate": false
|
| 96 |
+
},
|
| 97 |
+
"rope_theta": 150000,
|
| 98 |
+
"router_aux_loss_coef": 0.9,
|
| 99 |
+
"sliding_window": 128,
|
| 100 |
+
"swiglu_limit": 7.0,
|
| 101 |
+
"tie_word_embeddings": false,
|
| 102 |
+
"use_cache": true,
|
| 103 |
+
"vocab_size": 201088
|
| 104 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/43f30d107729db6b760b2a90833f43ca44f68e6468ef64e131f71e2fa9f5b23f/f71907dd62c8b40e471f.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "openai/gpt-oss-20b",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GptOssForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": true,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"experts_per_token": 4,
|
| 12 |
+
"head_dim": 64,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2880,
|
| 15 |
+
"initial_context_length": 4096,
|
| 16 |
+
"initializer_range": 0.02,
|
| 17 |
+
"intermediate_size": 2880,
|
| 18 |
+
"layer_types": [
|
| 19 |
+
"sliding_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"sliding_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"sliding_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"sliding_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"sliding_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"sliding_attention",
|
| 42 |
+
"full_attention"
|
| 43 |
+
],
|
| 44 |
+
"max_position_embeddings": 131072,
|
| 45 |
+
"model_type": "gpt_oss",
|
| 46 |
+
"neuron": {
|
| 47 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 48 |
+
"batch_size": 1,
|
| 49 |
+
"capacity_factor": null,
|
| 50 |
+
"checkpoint_id": "openai/gpt-oss-20b",
|
| 51 |
+
"checkpoint_revision": "6cee5e81ee83917806bbde320786a8fb61efebee",
|
| 52 |
+
"continuous_batching": false,
|
| 53 |
+
"ep_degree": 1,
|
| 54 |
+
"fused_qkv": false,
|
| 55 |
+
"glu_mlp": true,
|
| 56 |
+
"local_ranks_size": 8,
|
| 57 |
+
"max_batch_size": 1,
|
| 58 |
+
"max_context_length": 4096,
|
| 59 |
+
"max_topk": 256,
|
| 60 |
+
"n_active_tokens": 4096,
|
| 61 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 62 |
+
"on_device_sampling": false,
|
| 63 |
+
"optimum_neuron_version": "0.4.5.dev1",
|
| 64 |
+
"output_logits": false,
|
| 65 |
+
"pp_degree": 1,
|
| 66 |
+
"sequence_length": 4096,
|
| 67 |
+
"speculation_length": 0,
|
| 68 |
+
"start_rank_id": 0,
|
| 69 |
+
"target": "trn1",
|
| 70 |
+
"torch_dtype": "bfloat16",
|
| 71 |
+
"tp_degree": 8
|
| 72 |
+
},
|
| 73 |
+
"num_attention_heads": 64,
|
| 74 |
+
"num_experts_per_tok": 4,
|
| 75 |
+
"num_hidden_layers": 24,
|
| 76 |
+
"num_key_value_heads": 8,
|
| 77 |
+
"num_local_experts": 32,
|
| 78 |
+
"output_router_logits": false,
|
| 79 |
+
"quantization_config": {
|
| 80 |
+
"modules_to_not_convert": [
|
| 81 |
+
"model.layers.*.self_attn",
|
| 82 |
+
"model.layers.*.mlp.router",
|
| 83 |
+
"model.embed_tokens",
|
| 84 |
+
"lm_head"
|
| 85 |
+
],
|
| 86 |
+
"quant_method": "mxfp4"
|
| 87 |
+
},
|
| 88 |
+
"rms_norm_eps": 1e-05,
|
| 89 |
+
"rope_scaling": {
|
| 90 |
+
"beta_fast": 32.0,
|
| 91 |
+
"beta_slow": 1.0,
|
| 92 |
+
"factor": 32.0,
|
| 93 |
+
"original_max_position_embeddings": 4096,
|
| 94 |
+
"rope_type": "yarn",
|
| 95 |
+
"truncate": false
|
| 96 |
+
},
|
| 97 |
+
"rope_theta": 150000,
|
| 98 |
+
"router_aux_loss_coef": 0.9,
|
| 99 |
+
"sliding_window": 128,
|
| 100 |
+
"swiglu_limit": 7.0,
|
| 101 |
+
"tie_word_embeddings": false,
|
| 102 |
+
"use_cache": true,
|
| 103 |
+
"vocab_size": 201088
|
| 104 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev1/gpt_oss/openai/gpt-oss-20b/b76d86cc7c31e0e29c99.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "openai/gpt-oss-20b",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GptOssForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": true,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"experts_per_token": 4,
|
| 12 |
+
"head_dim": 64,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2880,
|
| 15 |
+
"initial_context_length": 4096,
|
| 16 |
+
"initializer_range": 0.02,
|
| 17 |
+
"intermediate_size": 2880,
|
| 18 |
+
"layer_types": [
|
| 19 |
+
"sliding_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"sliding_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"sliding_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"sliding_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"sliding_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"sliding_attention",
|
| 42 |
+
"full_attention"
|
| 43 |
+
],
|
| 44 |
+
"max_position_embeddings": 131072,
|
| 45 |
+
"model_type": "gpt_oss",
|
| 46 |
+
"neuron": {
|
| 47 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 48 |
+
"batch_size": 1,
|
| 49 |
+
"capacity_factor": null,
|
| 50 |
+
"checkpoint_id": "openai/gpt-oss-20b",
|
| 51 |
+
"checkpoint_revision": "6cee5e81ee83917806bbde320786a8fb61efebee",
|
| 52 |
+
"continuous_batching": false,
|
| 53 |
+
"ep_degree": 1,
|
| 54 |
+
"fused_qkv": true,
|
| 55 |
+
"glu_mlp": true,
|
| 56 |
+
"local_ranks_size": 8,
|
| 57 |
+
"max_batch_size": 1,
|
| 58 |
+
"max_context_length": 4096,
|
| 59 |
+
"max_topk": 256,
|
| 60 |
+
"n_active_tokens": 4096,
|
| 61 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 62 |
+
"on_device_sampling": true,
|
| 63 |
+
"optimum_neuron_version": "0.4.5.dev1",
|
| 64 |
+
"output_logits": false,
|
| 65 |
+
"pp_degree": 1,
|
| 66 |
+
"sequence_length": 4096,
|
| 67 |
+
"speculation_length": 0,
|
| 68 |
+
"start_rank_id": 0,
|
| 69 |
+
"target": "trn1",
|
| 70 |
+
"torch_dtype": "bfloat16",
|
| 71 |
+
"tp_degree": 8
|
| 72 |
+
},
|
| 73 |
+
"num_attention_heads": 64,
|
| 74 |
+
"num_experts_per_tok": 4,
|
| 75 |
+
"num_hidden_layers": 24,
|
| 76 |
+
"num_key_value_heads": 8,
|
| 77 |
+
"num_local_experts": 32,
|
| 78 |
+
"output_router_logits": false,
|
| 79 |
+
"quantization_config": {
|
| 80 |
+
"modules_to_not_convert": [
|
| 81 |
+
"model.layers.*.self_attn",
|
| 82 |
+
"model.layers.*.mlp.router",
|
| 83 |
+
"model.embed_tokens",
|
| 84 |
+
"lm_head"
|
| 85 |
+
],
|
| 86 |
+
"quant_method": "mxfp4"
|
| 87 |
+
},
|
| 88 |
+
"rms_norm_eps": 1e-05,
|
| 89 |
+
"rope_scaling": {
|
| 90 |
+
"beta_fast": 32.0,
|
| 91 |
+
"beta_slow": 1.0,
|
| 92 |
+
"factor": 32.0,
|
| 93 |
+
"original_max_position_embeddings": 4096,
|
| 94 |
+
"rope_type": "yarn",
|
| 95 |
+
"truncate": false
|
| 96 |
+
},
|
| 97 |
+
"rope_theta": 150000,
|
| 98 |
+
"router_aux_loss_coef": 0.9,
|
| 99 |
+
"sliding_window": 128,
|
| 100 |
+
"swiglu_limit": 7.0,
|
| 101 |
+
"tie_word_embeddings": false,
|
| 102 |
+
"use_cache": true,
|
| 103 |
+
"vocab_size": 201088
|
| 104 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/0cbac419f8839d90c6bafb19e17441f1a052e93e227f0fa62918ebe7d882e225/88fa555c8659e3f7fc5a.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 32,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": true,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 32,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": true,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/009b59fa3cc87705bbb4.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/06dea0fe4ee55a8035aa.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 16,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 32,
|
| 68 |
+
"max_batch_size": 16,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 32
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/0ccfeb749c49d4002400.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 64,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 64,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/0d6dd8f35029b2597ea3.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 32,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 32,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/12054fb3f8fa1b1fc9a6.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 32768,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 32768,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 32768,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/22e7cd94e28b353a796b.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/26b9c01e8f46a57f8933.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 16,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 16,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/2f14fea94ed53b3b1d94.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 16,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 32,
|
| 68 |
+
"max_batch_size": 16,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 32
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/34bc113dd21cb74d6179.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/44994da16f213a28fa4c.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 16,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 16,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/49052c1d22e88f219887.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 32768,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 32768,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 32768,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/49fc8d2feaf8eca91a07.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/535e7795fb656220e11a.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/55d254e5fde0dcab0e41.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 64,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 64,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/5e301b3a72a832a33468.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/644bf9550db2b89192aa.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/6501938b047f7d373cdd.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/67070f0a4d500338d5aa.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 32768,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 32768,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 32768,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/7bfaacc5ae3961c38f6c.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/824782edf021538e230f.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 64,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 64,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/8c7328c05cd751a24e18.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 128,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 128,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/8fdc4765f723aa1ce54e.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/912ca3353189929b7b15.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/99d0c0f90ad212bff9e2.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/99d672cbaa2e018da4e4.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/9d23ffa4ccbadb0a623e.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/9e94643f5e1e669914b1.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a41e0220fcb4438b8502.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 32768,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 32768,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 32768,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a45b5270d576ce1c1a2a.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 64,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 64,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a52b33ce4a3bb2a7e4bb.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/a8ca7ace639199dfc385.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 32,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 32,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/ad354086b250f133c9c6.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/aec994126b22dffefd0b.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 32,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 32,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b0bfc6ba654a35354148.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 64,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 64,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b2f13b35b2326e133272.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b4a6b1d49ffe1fb09037.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b52dd6b442b63b75e7b7.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/b8a28a3ac7bcba98b595.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/bba41446ac8406c873ee.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 32768,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 32768,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 32768,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/c2f89131f4ebe4bff600.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 128,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 8,
|
| 68 |
+
"max_batch_size": 128,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 8
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/c67e5e7f18f7dbf76e0c.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 16384,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 16384,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 16384,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/cade97aae05512df69b7.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 32,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 16,
|
| 68 |
+
"max_batch_size": 32,
|
| 69 |
+
"max_context_length": 2048,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 2048,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 2048,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 16
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d0e85bdeabc9387b9465.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 16,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 16,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d27d49077cbb0bf50eb9.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/af58eb15d8e02338dc2f2e880e9c6ec803a98278914b3606acdcc252e7e18429/d59c183509c2a322fd8b.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-Embedding-8B",
|
| 4 |
+
"_task": "feature-extraction",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 4096,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 12288,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"full_attention"
|
| 53 |
+
],
|
| 54 |
+
"max_position_embeddings": 40960,
|
| 55 |
+
"max_window_layers": 36,
|
| 56 |
+
"model_type": "qwen3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 8,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "Qwen/Qwen3-Embedding-8B",
|
| 62 |
+
"checkpoint_revision": "1d8ad4ca9b3dd8059ad90a75d4983776a23d44af",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 8,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": false,
|
| 74 |
+
"optimum_neuron_version": "0.4.5.dev2",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"num_attention_heads": 32,
|
| 85 |
+
"num_hidden_layers": 36,
|
| 86 |
+
"num_key_value_heads": 8,
|
| 87 |
+
"rms_norm_eps": 1e-06,
|
| 88 |
+
"rope_scaling": null,
|
| 89 |
+
"rope_theta": 1000000,
|
| 90 |
+
"sliding_window": null,
|
| 91 |
+
"tie_word_embeddings": false,
|
| 92 |
+
"use_cache": true,
|
| 93 |
+
"use_sliding_window": false,
|
| 94 |
+
"vocab_size": 151665
|
| 95 |
+
}
|