deVision v0.2 (github model-v0.2, 5437ac96b022e0c7029c40100575733dc852a707)
Browse files- .gitattributes +1 -34
- LICENSE +202 -0
- MANIFEST.json +50 -0
- README.md +212 -126
- config.json +19 -0
- encoder/config.json +84 -0
- evaluation/results.json +153 -0
- evaluation/results.md +37 -0
- model.safetensors +3 -0
- provenance.json +58 -0
- tokenizer/tokenizer.json +0 -0
- tokenizer/tokenizer_config.json +17 -0
- vision/config.json +15 -0
.gitattributes
CHANGED
|
@@ -1,35 +1,2 @@
|
|
| 1 |
-
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
-
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
-
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright [yyyy] [name of copyright owner]
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
MANIFEST.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
".gitattributes": {
|
| 3 |
+
"size": 92,
|
| 4 |
+
"sha256": "127386e47ae53f224aaf02383189e1b11e6d52289775b426655aa5d87d7e953c"
|
| 5 |
+
},
|
| 6 |
+
"LICENSE": {
|
| 7 |
+
"size": 11358,
|
| 8 |
+
"sha256": "cfc7749b96f63bd31c3c42b5c471bf756814053e847c10f3eb003417bc523d30"
|
| 9 |
+
},
|
| 10 |
+
"README.md": {
|
| 11 |
+
"size": 13520,
|
| 12 |
+
"sha256": "584f165e73880275bb92348107839e7483a012ef3fa23ea7a6717576e97c05c4"
|
| 13 |
+
},
|
| 14 |
+
"config.json": {
|
| 15 |
+
"size": 353,
|
| 16 |
+
"sha256": "26fdd9541256d09d8017bd467962bbf74a7cfc3986ca340e89e63faae3c8d11e"
|
| 17 |
+
},
|
| 18 |
+
"encoder/config.json": {
|
| 19 |
+
"size": 2084,
|
| 20 |
+
"sha256": "5268d24ad3b77c8151de5dcb0762ba4391619aad9ab0bda33e36fb083cfeae6d"
|
| 21 |
+
},
|
| 22 |
+
"evaluation/results.json": {
|
| 23 |
+
"size": 2819,
|
| 24 |
+
"sha256": "adb3ec9aeff1dc30974a9f9e2ed54c589672bb65b1d9b66e70792158172a9988"
|
| 25 |
+
},
|
| 26 |
+
"evaluation/results.md": {
|
| 27 |
+
"size": 1623,
|
| 28 |
+
"sha256": "cdfcdfe16977c52129a0fd7e9001fe7143bdbc494c9246ef92321aef2b9d645c"
|
| 29 |
+
},
|
| 30 |
+
"model.safetensors": {
|
| 31 |
+
"size": 2073754784,
|
| 32 |
+
"sha256": "07a3e8aacb986396d3975d12d672526c6d3b8ec240b9d9d503ef235c5c87c3ae"
|
| 33 |
+
},
|
| 34 |
+
"provenance.json": {
|
| 35 |
+
"size": 1324,
|
| 36 |
+
"sha256": "94d064129af62edafbcf05396b0a0b589034218f829070967039e8a3126e152c"
|
| 37 |
+
},
|
| 38 |
+
"tokenizer/tokenizer.json": {
|
| 39 |
+
"size": 3583228,
|
| 40 |
+
"sha256": "6c8aaa9a542084f2457eab775d4eeb51f92a70c0fd9de28d5edb0ddec3c08d30"
|
| 41 |
+
},
|
| 42 |
+
"tokenizer/tokenizer_config.json": {
|
| 43 |
+
"size": 379,
|
| 44 |
+
"sha256": "5926e6ec4294f80294bd98d9176defa9fbae7f140525d06a1da987488e74a973"
|
| 45 |
+
},
|
| 46 |
+
"vision/config.json": {
|
| 47 |
+
"size": 361,
|
| 48 |
+
"sha256": "118c80de2a3e1c4cd8142cfcc61dc1d55bb02013b2f9f0d6ca4a94cd857f77e4"
|
| 49 |
+
}
|
| 50 |
+
}
|
README.md
CHANGED
|
@@ -1,27 +1,153 @@
|
|
| 1 |
---
|
| 2 |
language:
|
| 3 |
- en
|
|
|
|
| 4 |
library_name: devision
|
| 5 |
pipeline_tag: visual-question-answering
|
| 6 |
base_model:
|
| 7 |
- convaiinnovations/laya
|
| 8 |
- google/siglip2-base-patch16-256
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 9 |
tags:
|
| 10 |
- devision
|
|
|
|
|
|
|
|
|
|
| 11 |
- system-one
|
| 12 |
-
- calibrated-decisions
|
| 13 |
- rlcd
|
| 14 |
-
-
|
| 15 |
-
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
---
|
| 17 |
|
| 18 |
-
# deVision
|
| 19 |
|
| 20 |
**Image + typed questions → structured answers and calibrated probabilities.** deVision is a non-autoregressive visual decision model built from Laya's ModernBERT-large decision model and a SigLIP2 vision encoder. It scores the supplied options without generating text. Multiple questions about one image share a single image encoding.
|
| 21 |
|
| 22 |
-
This
|
| 23 |
-
|
| 24 |
-
**Repository status, October 6, 2026:** this repository currently contains the model card; the V11 weights and configuration files have not yet been uploaded. Loading `lukbit/devision` will work after those files are added. With a local checkout and checkpoint, use `runs/v11-v9b-recovery/stage2` instead.
|
| 25 |
|
| 26 |
## Install
|
| 27 |
|
|
@@ -29,26 +155,19 @@ This card describes **V11 (`v11-v9b-recovery`)**, selected by the project owner
|
|
| 29 |
pip install "devision @ git+https://github.com/byebyebruce/devision"
|
| 30 |
```
|
| 31 |
|
| 32 |
-
Python 3.11 or newer is required
|
| 33 |
-
|
| 34 |
-
The Python examples below require the new `image=` API (code commit `f34e81a` or later). As of October 6, the GitHub default branch has not yet received this change. Until it does, install from an updated local deVision checkout with `pip install .`; installing from GitHub currently provides the legacy `state` image-part API.
|
| 35 |
|
| 36 |
## Python quickstart
|
| 37 |
|
| 38 |
-
The following Hub example applies once the weights are uploaded:
|
| 39 |
-
|
| 40 |
```python
|
| 41 |
import devision
|
| 42 |
|
| 43 |
-
model = devision.load("lukbit/devision", device="cpu")
|
| 44 |
result = model.predict(
|
| 45 |
-
image="photo.jpg", #
|
| 46 |
-
state="", #
|
| 47 |
questions={
|
| 48 |
-
"has_fork": {
|
| 49 |
-
"type": "noul",
|
| 50 |
-
"instructions": "Is there a fork in the image?",
|
| 51 |
-
},
|
| 52 |
"room": {
|
| 53 |
"type": "choice",
|
| 54 |
"instructions": "Which room is this?",
|
|
@@ -58,171 +177,138 @@ result = model.predict(
|
|
| 58 |
)
|
| 59 |
print(result["answers"]["has_fork"]["noul"]) # probability of yes
|
| 60 |
print(result["answers"]["room"]["choice"]) # selected option
|
| 61 |
-
print(result["answers"]["room"]["probabilities"]) # probability
|
| 62 |
```
|
| 63 |
|
| 64 |
-
`model.decide(...)`
|
| 65 |
|
| 66 |
### Image input
|
| 67 |
|
| 68 |
-
|
| 69 |
|
| 70 |
| Input | Example |
|
| 71 |
|---|---|
|
| 72 |
-
| HTTP(S) URL | `image="https://example.com/photo.jpg"`
|
| 73 |
-
| Local path | `image="photo.jpg"` or `image=Path("photo.jpg")`
|
| 74 |
-
| Base64 data URI | `image="data:image/png;base64,..."`
|
| 75 |
-
| Plain
|
| 76 |
-
| Encoded image bytes | `image=image_bytes`
|
| 77 |
| PIL image | `image=pil_image` |
|
| 78 |
|
| 79 |
-
### Context
|
| 80 |
|
| 81 |
-
`state`
|
| 82 |
|
| 83 |
```python
|
| 84 |
result = model.decide(
|
| 85 |
image="photo.jpg",
|
| 86 |
state={"note": "The customer says this item arrived damaged."},
|
| 87 |
-
questions={
|
| 88 |
-
"damaged": {
|
| 89 |
-
"type": "noul",
|
| 90 |
-
"instructions": "Is the item visibly damaged?",
|
| 91 |
-
},
|
| 92 |
-
},
|
| 93 |
)
|
| 94 |
```
|
| 95 |
|
| 96 |
-
|
| 97 |
|
| 98 |
### Output
|
| 99 |
|
| 100 |
-
- `noul`
|
| 101 |
-
- `choice`
|
| 102 |
-
-
|
| 103 |
-
- Invalid
|
| 104 |
|
| 105 |
## HTTP API and browser demo
|
| 106 |
|
| 107 |
-
Once the weights are uploaded:
|
| 108 |
-
|
| 109 |
```bash
|
| 110 |
-
pip install "devision
|
| 111 |
devision-serve --checkpoint lukbit/devision --device cpu --port 8000
|
| 112 |
```
|
| 113 |
|
| 114 |
-
Open `http://127.0.0.1:8000` for the demo.
|
| 115 |
-
|
| 116 |
-
HTTP keeps the Jev-compatible `/v1/systemone` format. The following example uploads a local image as Base64:
|
| 117 |
|
| 118 |
-
```
|
| 119 |
-
|
| 120 |
-
|
| 121 |
-
from pathlib import Path
|
| 122 |
-
from urllib.request import Request, urlopen
|
| 123 |
-
|
| 124 |
-
body = {
|
| 125 |
-
"state": [{
|
| 126 |
-
"type": "image",
|
| 127 |
-
"base64": base64.b64encode(Path("photo.jpg").read_bytes()).decode("ascii"),
|
| 128 |
-
}],
|
| 129 |
-
"questions": {
|
| 130 |
-
"has_fork": {"type": "noul", "instructions": "Is there a fork in the image?"},
|
| 131 |
-
},
|
| 132 |
-
}
|
| 133 |
-
request = Request(
|
| 134 |
-
"http://127.0.0.1:8000/v1/systemone",
|
| 135 |
-
data=json.dumps(body).encode("utf-8"),
|
| 136 |
-
headers={"Content-Type": "application/json"},
|
| 137 |
-
)
|
| 138 |
-
with urlopen(request, timeout=30) as response:
|
| 139 |
-
print(json.load(response))
|
| 140 |
```
|
| 141 |
|
| 142 |
-
|
| 143 |
|
| 144 |
## Architecture
|
| 145 |
|
| 146 |
-
| Component |
|
| 147 |
|---|---|
|
| 148 |
| Vision encoder | Frozen SigLIP2-B/16, 256 × 256 input, about 93M parameters |
|
| 149 |
| Image preprocessing | Aspect-preserving resize and letterbox padding |
|
| 150 |
-
| Connector | 2 × 2 patch grouping
|
| 151 |
-
| Visual sequence |
|
| 152 |
-
| Text encoder | Laya-
|
| 153 |
| Decision head | Laya's two Transformer layers and option scorer, about 27M parameters |
|
| 154 |
-
| Checkpoint | About 518M parameters, FP32,
|
| 155 |
|
| 156 |
-
Each option is scored at its `[MASK]` position
|
| 157 |
|
| 158 |
## Training and calibration
|
| 159 |
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
- 3,000 counting questions;
|
| 163 |
-
- 4,500 replay questions, including ScienceQA, VSR and Visual7W;
|
| 164 |
-
- 2,488 left/right questions retained from V9B.
|
| 165 |
|
| 166 |
-
|
| 167 |
|
| 168 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 169 |
|
| 170 |
-
|
| 171 |
-
|---|---|
|
| 172 |
-
| Choice, 2 options | 2.1514 |
|
| 173 |
-
| Choice, 3–5 options | 1.9454 |
|
| 174 |
-
| Choice fallback | 1.9879 |
|
| 175 |
-
| Noul | 1.8161 |
|
| 176 |
-
|
| 177 |
-
Calibration changes probabilities, not the model's underlying visual ability. Validate confidence thresholds on your own distribution.
|
| 178 |
|
| 179 |
## Evaluation
|
| 180 |
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
| Set | Questions | Accuracy | ECE |
|
| 184 |
-
|---|---:|---:|---:|
|
| 185 |
-
| COCO object presence | 1,120 | 0.
|
| 186 |
-
| VQAv2 multiple choice | 1,420 | 0.
|
| 187 |
-
| COCO size | 1,100 | 0.
|
| 188 |
-
| POPE, project filtered set | 8,676 | 0.
|
| 189 |
-
| GQA val subset | 992 | 0.
|
| 190 |
-
| COCO position | 1,274 | 0.
|
| 191 |
-
| VQAv2 yes/no subset | 1,000 | 0.
|
| 192 |
-
| COCO relative position | 1,950 | 0.
|
| 193 |
-
| VSR, project held-out split | 904 | 0.
|
| 194 |
-
| Visual7W, project held-out split | 1,000 | 0.
|
| 195 |
-
| Fresh counting test | 600 | 0.
|
| 196 |
|
| 197 |
### Comparison with Laya Vision 201M
|
| 198 |
|
| 199 |
-
|
| 200 |
|
| 201 |
-
| Set | Questions | Laya Vision | deVision
|
| 202 |
|---|---:|---:|---:|---|
|
| 203 |
-
| VQAv2 yes/no | 4,887 | 0.717 | 0.
|
| 204 |
-
| A-OKVQA | 1,138 | 0.598 | 0.
|
| 205 |
-
| ScienceQA with images | 2,097 | 0.824 | 0.
|
|
|
|
|
|
|
|
|
|
| 206 |
|
| 207 |
-
|
| 208 |
|
| 209 |
-
|
| 210 |
|
| 211 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 212 |
|
| 213 |
-
|
| 214 |
-
- **Spatial reasoning is incomplete.** Correct answers on both an original image and its mirror occur for about 61% of left/right position pairs and 39% of relative-position choice pairs, but only 10% of relative-position yes/no pairs.
|
| 215 |
-
- **Scientific diagrams remain difficult.** On 323 natural-science questions judged to require the image, V11 scores 0.467 versus Laya Vision's 0.700. On the full ScienceQA set, swapping in unrelated images still yields 0.665 accuracy, compared with 0.747 for the correct images; text and answer priors contribute substantially.
|
| 216 |
-
- **Old-task protection is uncertain in two comparisons.** VSR and A-OKVQA did not establish the predefined lower bound of −2 percentage points relative to V9B. This is uncertainty about protection, not proof of a significant decline.
|
| 217 |
-
- **Small text and fine detail are limited by 256 × 256 input.** The model is not a general-purpose OCR or text-generation system.
|
| 218 |
-
- **English, single image, `noul` and `choice` only.** Other languages, multi-image input and `score` questions are not supported.
|
| 219 |
-
- **Probabilities can be miscalibrated out of domain.** A low ECE on one benchmark does not guarantee reliable confidence elsewhere.
|
| 220 |
|
| 221 |
-
|
| 222 |
|
| 223 |
-
|
| 224 |
-
- [Laya decision model](https://huggingface.co/convaiinnovations/laya)
|
| 225 |
-
- [SigLIP2 vision encoder](https://huggingface.co/google/siglip2-base-patch16-256)
|
| 226 |
-
- [Laya Vision reference implementation and published evaluations](https://github.com/r33drichards/laya-vision)
|
| 227 |
|
| 228 |
-
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
language:
|
| 3 |
- en
|
| 4 |
+
license: apache-2.0
|
| 5 |
library_name: devision
|
| 6 |
pipeline_tag: visual-question-answering
|
| 7 |
base_model:
|
| 8 |
- convaiinnovations/laya
|
| 9 |
- google/siglip2-base-patch16-256
|
| 10 |
+
datasets:
|
| 11 |
+
- HuggingFaceM4/VQAv2
|
| 12 |
+
- lmms-lab/GQA
|
| 13 |
+
- HuggingFaceM4/A-OKVQA
|
| 14 |
+
- derek-thomas/ScienceQA
|
| 15 |
+
- HuggingFaceM4/the_cauldron
|
| 16 |
+
- cambridgeltl/vsr_random
|
| 17 |
+
- jxu124/objects365
|
| 18 |
+
- J1mb0o/e-SNLI-VE
|
| 19 |
+
- Vision-Flan/vision-flan_191-task_1k
|
| 20 |
+
- ryokamoi/VisOnlyQA_Train
|
| 21 |
+
- RyanWW/Super-CLEVR
|
| 22 |
+
- allenai/pixmo-count
|
| 23 |
tags:
|
| 24 |
- devision
|
| 25 |
+
- visual-decisions
|
| 26 |
+
- calibrated-probabilities
|
| 27 |
+
- jev
|
| 28 |
- system-one
|
|
|
|
| 29 |
- rlcd
|
| 30 |
+
- cpu
|
| 31 |
+
model-index:
|
| 32 |
+
- name: deVision v0.2
|
| 33 |
+
results:
|
| 34 |
+
- task:
|
| 35 |
+
type: visual-question-answering
|
| 36 |
+
dataset:
|
| 37 |
+
name: COCO object presence
|
| 38 |
+
type: test_exist
|
| 39 |
+
metrics:
|
| 40 |
+
- type: accuracy
|
| 41 |
+
value: 0.9384
|
| 42 |
+
- type: ece
|
| 43 |
+
value: 0.0204
|
| 44 |
+
- task:
|
| 45 |
+
type: visual-question-answering
|
| 46 |
+
dataset:
|
| 47 |
+
name: VQAv2 multiple choice
|
| 48 |
+
type: test_vqa_choice
|
| 49 |
+
metrics:
|
| 50 |
+
- type: accuracy
|
| 51 |
+
value: 0.8979
|
| 52 |
+
- type: ece
|
| 53 |
+
value: 0.0262
|
| 54 |
+
- task:
|
| 55 |
+
type: visual-question-answering
|
| 56 |
+
dataset:
|
| 57 |
+
name: COCO size
|
| 58 |
+
type: test_size
|
| 59 |
+
metrics:
|
| 60 |
+
- type: accuracy
|
| 61 |
+
value: 0.86
|
| 62 |
+
- type: ece
|
| 63 |
+
value: 0.036
|
| 64 |
+
- task:
|
| 65 |
+
type: visual-question-answering
|
| 66 |
+
dataset:
|
| 67 |
+
name: POPE, project filtered set
|
| 68 |
+
type: bench_pope
|
| 69 |
+
metrics:
|
| 70 |
+
- type: accuracy
|
| 71 |
+
value: 0.8511
|
| 72 |
+
- type: ece
|
| 73 |
+
value: 0.0578
|
| 74 |
+
- task:
|
| 75 |
+
type: visual-question-answering
|
| 76 |
+
dataset:
|
| 77 |
+
name: GQA val subset
|
| 78 |
+
type: test_gqa
|
| 79 |
+
metrics:
|
| 80 |
+
- type: accuracy
|
| 81 |
+
value: 0.7843
|
| 82 |
+
- type: ece
|
| 83 |
+
value: 0.0419
|
| 84 |
+
- task:
|
| 85 |
+
type: visual-question-answering
|
| 86 |
+
dataset:
|
| 87 |
+
name: COCO position
|
| 88 |
+
type: test_position
|
| 89 |
+
metrics:
|
| 90 |
+
- type: accuracy
|
| 91 |
+
value: 0.8721
|
| 92 |
+
- type: ece
|
| 93 |
+
value: 0.0251
|
| 94 |
+
- task:
|
| 95 |
+
type: visual-question-answering
|
| 96 |
+
dataset:
|
| 97 |
+
name: VQAv2 yes/no subset
|
| 98 |
+
type: test_vqa_yesno
|
| 99 |
+
metrics:
|
| 100 |
+
- type: accuracy
|
| 101 |
+
value: 0.71
|
| 102 |
+
- type: ece
|
| 103 |
+
value: 0.0223
|
| 104 |
+
- task:
|
| 105 |
+
type: visual-question-answering
|
| 106 |
+
dataset:
|
| 107 |
+
name: COCO relative position
|
| 108 |
+
type: test_relation
|
| 109 |
+
metrics:
|
| 110 |
+
- type: accuracy
|
| 111 |
+
value: 0.7108
|
| 112 |
+
- type: ece
|
| 113 |
+
value: 0.0471
|
| 114 |
+
- task:
|
| 115 |
+
type: visual-question-answering
|
| 116 |
+
dataset:
|
| 117 |
+
name: VSR, project held-out split
|
| 118 |
+
type: test_vsr
|
| 119 |
+
metrics:
|
| 120 |
+
- type: accuracy
|
| 121 |
+
value: 0.6792
|
| 122 |
+
- type: ece
|
| 123 |
+
value: 0.0478
|
| 124 |
+
- task:
|
| 125 |
+
type: visual-question-answering
|
| 126 |
+
dataset:
|
| 127 |
+
name: Visual7W, project held-out split
|
| 128 |
+
type: test_v7w
|
| 129 |
+
metrics:
|
| 130 |
+
- type: accuracy
|
| 131 |
+
value: 0.721
|
| 132 |
+
- type: ece
|
| 133 |
+
value: 0.0302
|
| 134 |
+
- task:
|
| 135 |
+
type: visual-question-answering
|
| 136 |
+
dataset:
|
| 137 |
+
name: Fresh counting test (unseen pictures)
|
| 138 |
+
type: test_count_fresh
|
| 139 |
+
metrics:
|
| 140 |
+
- type: accuracy
|
| 141 |
+
value: 0.74
|
| 142 |
+
- type: ece
|
| 143 |
+
value: 0.0648
|
| 144 |
---
|
| 145 |
|
| 146 |
+
# deVision v0.2
|
| 147 |
|
| 148 |
**Image + typed questions → structured answers and calibrated probabilities.** deVision is a non-autoregressive visual decision model built from Laya's ModernBERT-large decision model and a SigLIP2 vision encoder. It scores the supplied options without generating text. Multiple questions about one image share a single image encoding.
|
| 149 |
|
| 150 |
+
This is release **v0.2**. It supports English, one image per request, yes/no decisions (`noul`) and multiple-choice decisions (`choice`). Weights and code are released under the Apache-2.0 licence; see [Licence and data](#licence-and-data).
|
|
|
|
|
|
|
| 151 |
|
| 152 |
## Install
|
| 153 |
|
|
|
|
| 155 |
pip install "devision @ git+https://github.com/byebyebruce/devision"
|
| 156 |
```
|
| 157 |
|
| 158 |
+
Python 3.11 or newer is required; the install includes the HTTP server and browser demo. If the model repository is private, authenticate first with `hf auth login` using an account that has access.
|
|
|
|
|
|
|
| 159 |
|
| 160 |
## Python quickstart
|
| 161 |
|
|
|
|
|
|
|
| 162 |
```python
|
| 163 |
import devision
|
| 164 |
|
| 165 |
+
model = devision.load("lukbit/devision", revision="v0.2", device="cpu")
|
| 166 |
result = model.predict(
|
| 167 |
+
image="photo.jpg", # an image path, an http(s) URL, base64, bytes or a PIL image
|
| 168 |
+
state="", # text context; empty when there is none
|
| 169 |
questions={
|
| 170 |
+
"has_fork": {"type": "noul", "instructions": "Is there a fork in the image?"},
|
|
|
|
|
|
|
|
|
|
| 171 |
"room": {
|
| 172 |
"type": "choice",
|
| 173 |
"instructions": "Which room is this?",
|
|
|
|
| 177 |
)
|
| 178 |
print(result["answers"]["has_fork"]["noul"]) # probability of yes
|
| 179 |
print(result["answers"]["room"]["choice"]) # selected option
|
| 180 |
+
print(result["answers"]["room"]["probabilities"]) # probability of each option
|
| 181 |
```
|
| 182 |
|
| 183 |
+
`model.decide(...)` takes the same arguments. Pin `revision="v0.2"` so that later releases do not change your results. The default device is `auto` (CUDA, then MPS, then CPU).
|
| 184 |
|
| 185 |
### Image input
|
| 186 |
|
| 187 |
+
The Python `image=` argument accepts:
|
| 188 |
|
| 189 |
| Input | Example |
|
| 190 |
|---|---|
|
| 191 |
+
| HTTP(S) URL | `image="https://example.com/photo.jpg"` |
|
| 192 |
+
| Local path | `image="photo.jpg"` or `image=Path("photo.jpg")` |
|
| 193 |
+
| Base64 data URI | `image="data:image/png;base64,..."` |
|
| 194 |
+
| Plain base64 | `image=base64_string` |
|
| 195 |
+
| Encoded image bytes | `image=image_bytes` (PNG / JPEG file contents) |
|
| 196 |
| PIL image | `image=pil_image` |
|
| 197 |
|
| 198 |
+
### Context
|
| 199 |
|
| 200 |
+
`state` carries text context as a string, object or array, exactly as in Jev:
|
| 201 |
|
| 202 |
```python
|
| 203 |
result = model.decide(
|
| 204 |
image="photo.jpg",
|
| 205 |
state={"note": "The customer says this item arrived damaged."},
|
| 206 |
+
questions={"damaged": {"type": "noul", "instructions": "Is the item visibly damaged?"}},
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 207 |
)
|
| 208 |
```
|
| 209 |
|
| 210 |
+
An `image` key inside an object state stays text: `state={"image": "photo.jpg"}` does not load the file. The Jev-compatible image part `state=[{"type": "image", "url": "https://..."}]` also works; do not combine it with `image=`.
|
| 211 |
|
| 212 |
### Output
|
| 213 |
|
| 214 |
+
- `noul`: the probability that the statement holds, between 0 and 1.
|
| 215 |
+
- `choice`: the highest-probability option; `probabilities` covers every option and sums to 1.
|
| 216 |
+
- `confidence` (choice) follows Jev: `(n * p_max - 1) / (n - 1)`.
|
| 217 |
+
- Invalid requests raise `devision.InvalidRequest`.
|
| 218 |
|
| 219 |
## HTTP API and browser demo
|
| 220 |
|
|
|
|
|
|
|
| 221 |
```bash
|
| 222 |
+
pip install "devision @ git+https://github.com/byebyebruce/devision"
|
| 223 |
devision-serve --checkpoint lukbit/devision --device cpu --port 8000
|
| 224 |
```
|
| 225 |
|
| 226 |
+
Open `http://127.0.0.1:8000` for the demo. The server speaks the Jev-compatible `POST /v1/systemone` format; an image is an element of the `state` array:
|
|
|
|
|
|
|
| 227 |
|
| 228 |
+
```json
|
| 229 |
+
{"state": [{"type": "image", "url": "https://example.com/photo.jpg"}, "optional text"],
|
| 230 |
+
"questions": {"has_fork": {"type": "noul", "instructions": "Is there a fork in the image?"}}}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 231 |
```
|
| 232 |
|
| 233 |
+
Use `{"type": "image", "base64": "..."}` for an uploaded picture. HTTP never reads server-local paths. Invalid requests and `score` questions return HTTP 422.
|
| 234 |
|
| 235 |
## Architecture
|
| 236 |
|
| 237 |
+
| Component | |
|
| 238 |
|---|---|
|
| 239 |
| Vision encoder | Frozen SigLIP2-B/16, 256 × 256 input, about 93M parameters |
|
| 240 |
| Image preprocessing | Aspect-preserving resize and letterbox padding |
|
| 241 |
+
| Connector | 2 × 2 patch grouping, then LayerNorm → Linear → GELU → Linear |
|
| 242 |
+
| Visual sequence | 64 visual tokens, inserted after `[CLS]` |
|
| 243 |
+
| Text encoder | Laya-initialised ModernBERT-large, about 395M parameters |
|
| 244 |
| Decision head | Laya's two Transformer layers and option scorer, about 27M parameters |
|
| 245 |
+
| Checkpoint | About 518M parameters, FP32, 2.07 GB of weights (LoRA merged) |
|
| 246 |
|
| 247 |
+
Each option is scored at its `[MASK]` position; a temperature-scaled softmax turns the scores into probabilities.
|
| 248 |
|
| 249 |
## Training and calibration
|
| 250 |
|
| 251 |
+
deVision is trained in two stages. Stage 1 aligns the projector on COCO captions (image-conditioned masked words). Stage 2 trains the decision on yes / no and multiple-choice questions with Laya's RLCD objective (projector, decision head and a LoRA on ModernBERT; the LoRA is merged for release). Earlier training covered COCO-derived existence / position / size questions, VQAv2, GQA, A-OKVQA, ScienceQA, AI2D, TQA, VSR, Visual7W telling, mirrored left/right pairs and counting. The last training step continued from that checkpoint for one pass over 353,830 questions: an extension pack of 315,342 questions from thirteen public sources (Objects365, TallyQA, CLEVR, CLEVR-Math, Super-CLEVR, FigureQA, MapQA, IconQA, SNLI-VE, Vision-Flan, VisOnlyQA, SpatialSense, PixMo-Count) plus 38,488 replayed questions of the earlier abilities; 44,229 steps on Apple MPS in FP32, learning rates `5e-5` (projector, head) and `1e-4` (LoRA), 500 warm-up steps. Full configurations and results: the [training log](https://github.com/byebyebruce/devision/blob/master/docs/training-log.md).
|
|
|
|
|
|
|
|
|
|
|
|
|
| 252 |
|
| 253 |
+
Temperatures fitted on the project's calibration set (a question's type / option-count bucket first, else its type):
|
| 254 |
|
| 255 |
+
| Bucket | Temperature |
|
| 256 |
+
|---|---:|
|
| 257 |
+
| Choice, 2 options | 1.8836 |
|
| 258 |
+
| Choice, 3–5 options | 1.9790 |
|
| 259 |
+
| Noul (yes / no) | 1.6382 |
|
| 260 |
+
| Choice, any other count | 1.9568 |
|
| 261 |
|
| 262 |
+
Calibration changes probabilities, not visual ability. Validate confidence thresholds on your own data.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 263 |
|
| 264 |
## Evaluation
|
| 265 |
|
| 266 |
+
Results of this checkpoint through `decide`, with the fitted temperatures; ECE uses 15 bins. *Mismatched* is the accuracy when every picture is swapped for an unrelated one (how much the answer depends on the picture). The project test sets use held-out pictures; they are project-specific subsets or generated questions, not official leaderboard scores, and several have informed decisions across training rounds.
|
| 267 |
+
|
| 268 |
+
| Set | Questions | Accuracy | Mismatched | ECE |
|
| 269 |
+
|---|---:|---:|---:|---:|
|
| 270 |
+
| COCO object presence | 1,120 | 0.938 | 0.493 | 0.020 |
|
| 271 |
+
| VQAv2 multiple choice | 1,420 | 0.898 | 0.399 | 0.026 |
|
| 272 |
+
| COCO size | 1,100 | 0.860 | 0.504 | 0.036 |
|
| 273 |
+
| POPE, project filtered set | 8,676 | 0.851 | 0.527 | 0.058 |
|
| 274 |
+
| GQA val subset | 992 | 0.784 | 0.532 | 0.042 |
|
| 275 |
+
| COCO position | 1,274 | 0.872 | 0.493 | 0.025 |
|
| 276 |
+
| VQAv2 yes/no subset | 1,000 | 0.710 | 0.518 | 0.022 |
|
| 277 |
+
| COCO relative position | 1,950 | 0.711 | 0.484 | 0.047 |
|
| 278 |
+
| VSR, project held-out split | 904 | 0.679 | 0.481 | 0.048 |
|
| 279 |
+
| Visual7W, project held-out split | 1,000 | 0.721 | 0.422 | 0.030 |
|
| 280 |
+
| Fresh counting test (unseen pictures) | 600 | 0.740 | 0.507 | 0.065 |
|
| 281 |
|
| 282 |
### Comparison with Laya Vision 201M
|
| 283 |
|
| 284 |
+
Laya Vision's published per-question predictions on the same questions; differences in percentage points with 95% intervals from paired resampling by picture. We did not rerun Laya Vision.
|
| 285 |
|
| 286 |
+
| Set | Questions | Laya Vision | deVision | Difference |
|
| 287 |
|---|---:|---:|---:|---|
|
| 288 |
+
| VQAv2 yes/no | 4,887 | 0.717 | 0.725 | +0.8 [-0.9, +2.3] |
|
| 289 |
+
| A-OKVQA | 1,138 | 0.598 | 0.626 | +2.7 [-0.6, +6.0] |
|
| 290 |
+
| ScienceQA with images | 2,097 | 0.824 | 0.766 | -5.8 [-7.9, -3.6] |
|
| 291 |
+
| ScienceQA natural science, needs the picture | 323 | 0.700 | 0.455 | -24.5 [-31.6, -17.6] |
|
| 292 |
+
|
| 293 |
+
On the full POPE random / popular / adversarial sets (3,000 questions each) deVision scores **0.891 / 0.868 / 0.791**, against Laya Vision's published **0.836 / 0.819 / 0.777** (aggregate scores only, not paired).
|
| 294 |
|
| 295 |
+
CPU latency: P50 148 ms, P95 161 ms (Darwin arm64, 4 threads, FP32, one question per request, warm-up excluded). Several questions about one picture in one request share the image encoding, so each extra question costs less.
|
| 296 |
|
| 297 |
+
## Limitations
|
| 298 |
|
| 299 |
+
- **Scientific diagrams and charts remain difficult.** On the 323 ScienceQA natural-science questions that need the picture, this release is about 24 points behind Laya Vision (see the comparison table). Comparing two named regions of a picture (two magnet poles, two series of a chart, two regions of a map, two line segments) is the main open weakness; training on tens of thousands of such questions left it near chance.
|
| 300 |
+
- **Spatial reasoning is incomplete.** A left/right question is answered right on both a picture and its mirror for about 72% of position pairs and 65% of relative-position choice pairs, but only 19% of relative-position yes / no pairs.
|
| 301 |
+
- **Object presence leans towards "yes".** Compared with the previous checkpoint, it more often says an object is present when the annotation says it is not (COCO existence test 95.5% → 93.8%; POPE adversarial 79.1%).
|
| 302 |
+
- **Small text and fine detail** are limited by the 256 × 256 input; this is not an OCR model.
|
| 303 |
+
- **English, one image, `noul` and `choice` only.** Other languages, several images and `score` questions are not supported.
|
| 304 |
+
- **Probabilities can be miscalibrated out of domain.** A low ECE on these benchmarks does not guarantee reliable confidence on your data.
|
| 305 |
|
| 306 |
+
## Licence and data
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 307 |
|
| 308 |
+
The weights and the code are released under the **Apache-2.0** licence, the licence of the three base models ([Laya](https://huggingface.co/convaiinnovations/laya), [SigLIP2](https://huggingface.co/google/siglip2-base-patch16-256), [ModernBERT](https://huggingface.co/answerdotai/ModernBERT-large)). The training data come from public datasets with their own terms, some of them non-commercial (for example ScienceQA, CC BY-NC-SA 4.0) or covering the pictures separately (COCO / Flickr images). Whether such terms carry over to trained weights is not settled; check the datasets listed in this card's metadata against your use.
|
| 309 |
|
| 310 |
+
## Links
|
|
|
|
|
|
|
|
|
|
| 311 |
|
| 312 |
+
- [Code, training configurations and experiment records](https://github.com/byebyebruce/devision) — this release is tag `model-v0.2`
|
| 313 |
+
- [Evaluation details](evaluation/results.md) · [metrics](evaluation/results.json) · [provenance](provenance.json)
|
| 314 |
+
- [Laya decision model](https://huggingface.co/convaiinnovations/laya) · [SigLIP2](https://huggingface.co/google/siglip2-base-patch16-256) · [Laya Vision](https://github.com/r33drichards/laya-vision)
|
config.json
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model_name": "devision-v0.2",
|
| 3 |
+
"image_size": 256,
|
| 4 |
+
"visual_shuffle": 2,
|
| 5 |
+
"max_len": 512,
|
| 6 |
+
"head_max_len": 192,
|
| 7 |
+
"head_layers": 2,
|
| 8 |
+
"n_act": 2,
|
| 9 |
+
"temperature": [
|
| 10 |
+
1.9567745923995972,
|
| 11 |
+
1.0,
|
| 12 |
+
1.6382114887237549
|
| 13 |
+
],
|
| 14 |
+
"temperature_by_options": {
|
| 15 |
+
"choice:2": 1.8836495876312256,
|
| 16 |
+
"choice:3-5": 1.979034662246704,
|
| 17 |
+
"noul:2": 1.6382114887237549
|
| 18 |
+
}
|
| 19 |
+
}
|
encoder/config.json
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"ModernBertForMaskedLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"bos_token_id": 50281,
|
| 8 |
+
"classifier_activation": "gelu",
|
| 9 |
+
"classifier_bias": false,
|
| 10 |
+
"classifier_dropout": 0.0,
|
| 11 |
+
"classifier_pooling": "mean",
|
| 12 |
+
"cls_token_id": 50281,
|
| 13 |
+
"decoder_bias": true,
|
| 14 |
+
"deterministic_flash_attn": false,
|
| 15 |
+
"dtype": "float32",
|
| 16 |
+
"embedding_dropout": 0.0,
|
| 17 |
+
"eos_token_id": 50282,
|
| 18 |
+
"global_attn_every_n_layers": 3,
|
| 19 |
+
"gradient_checkpointing": false,
|
| 20 |
+
"hidden_activation": "gelu",
|
| 21 |
+
"hidden_size": 1024,
|
| 22 |
+
"initializer_cutoff_factor": 2.0,
|
| 23 |
+
"initializer_range": 0.02,
|
| 24 |
+
"intermediate_size": 2624,
|
| 25 |
+
"layer_norm_eps": 1e-05,
|
| 26 |
+
"layer_types": [
|
| 27 |
+
"full_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"sliding_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"sliding_attention",
|
| 38 |
+
"sliding_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"sliding_attention",
|
| 41 |
+
"sliding_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"sliding_attention",
|
| 44 |
+
"sliding_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"sliding_attention",
|
| 47 |
+
"sliding_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"sliding_attention",
|
| 50 |
+
"sliding_attention",
|
| 51 |
+
"full_attention",
|
| 52 |
+
"sliding_attention",
|
| 53 |
+
"sliding_attention",
|
| 54 |
+
"full_attention"
|
| 55 |
+
],
|
| 56 |
+
"local_attention": 128,
|
| 57 |
+
"max_position_embeddings": 8192,
|
| 58 |
+
"mlp_bias": false,
|
| 59 |
+
"mlp_dropout": 0.0,
|
| 60 |
+
"model_type": "modernbert",
|
| 61 |
+
"norm_bias": false,
|
| 62 |
+
"norm_eps": 1e-05,
|
| 63 |
+
"num_attention_heads": 16,
|
| 64 |
+
"num_hidden_layers": 28,
|
| 65 |
+
"pad_token_id": 50283,
|
| 66 |
+
"position_embedding_type": "absolute",
|
| 67 |
+
"repad_logits_with_grad": false,
|
| 68 |
+
"rope_parameters": {
|
| 69 |
+
"full_attention": {
|
| 70 |
+
"rope_theta": 160000.0,
|
| 71 |
+
"rope_type": "default"
|
| 72 |
+
},
|
| 73 |
+
"sliding_attention": {
|
| 74 |
+
"rope_theta": 10000.0,
|
| 75 |
+
"rope_type": "default"
|
| 76 |
+
}
|
| 77 |
+
},
|
| 78 |
+
"sep_token_id": 50282,
|
| 79 |
+
"sparse_pred_ignore_index": -100,
|
| 80 |
+
"sparse_prediction": false,
|
| 81 |
+
"tie_word_embeddings": true,
|
| 82 |
+
"transformers_version": "5.17.0",
|
| 83 |
+
"vocab_size": 50368
|
| 84 |
+
}
|
evaluation/results.json
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"temperature": {
|
| 3 |
+
"choice": 1.9567745923995972,
|
| 4 |
+
"noul": 1.6382114887237549,
|
| 5 |
+
"by_options": {
|
| 6 |
+
"choice:2": 1.8836495876312256,
|
| 7 |
+
"choice:3-5": 1.979034662246704,
|
| 8 |
+
"noul:2": 1.6382114887237549
|
| 9 |
+
}
|
| 10 |
+
},
|
| 11 |
+
"test_sets": {
|
| 12 |
+
"test_exist": {
|
| 13 |
+
"questions": 1120,
|
| 14 |
+
"accuracy": 0.9384,
|
| 15 |
+
"nll": 0.162,
|
| 16 |
+
"ece": 0.0204,
|
| 17 |
+
"accuracy_mismatched": 0.4929
|
| 18 |
+
},
|
| 19 |
+
"test_vqa_choice": {
|
| 20 |
+
"questions": 1420,
|
| 21 |
+
"accuracy": 0.8979,
|
| 22 |
+
"nll": 0.2706,
|
| 23 |
+
"ece": 0.0262,
|
| 24 |
+
"accuracy_mismatched": 0.3993
|
| 25 |
+
},
|
| 26 |
+
"test_size": {
|
| 27 |
+
"questions": 1100,
|
| 28 |
+
"accuracy": 0.86,
|
| 29 |
+
"nll": 0.3503,
|
| 30 |
+
"ece": 0.036,
|
| 31 |
+
"accuracy_mismatched": 0.5045
|
| 32 |
+
},
|
| 33 |
+
"bench_pope": {
|
| 34 |
+
"questions": 8676,
|
| 35 |
+
"accuracy": 0.8511,
|
| 36 |
+
"nll": 0.3702,
|
| 37 |
+
"ece": 0.0578,
|
| 38 |
+
"accuracy_mismatched": 0.527
|
| 39 |
+
},
|
| 40 |
+
"test_gqa": {
|
| 41 |
+
"questions": 992,
|
| 42 |
+
"accuracy": 0.7843,
|
| 43 |
+
"nll": 0.409,
|
| 44 |
+
"ece": 0.0419,
|
| 45 |
+
"accuracy_mismatched": 0.5323
|
| 46 |
+
},
|
| 47 |
+
"test_position": {
|
| 48 |
+
"questions": 1274,
|
| 49 |
+
"accuracy": 0.8721,
|
| 50 |
+
"nll": 0.3045,
|
| 51 |
+
"ece": 0.0251,
|
| 52 |
+
"accuracy_mismatched": 0.4929
|
| 53 |
+
},
|
| 54 |
+
"test_vqa_yesno": {
|
| 55 |
+
"questions": 1000,
|
| 56 |
+
"accuracy": 0.71,
|
| 57 |
+
"nll": 0.5768,
|
| 58 |
+
"ece": 0.0223,
|
| 59 |
+
"accuracy_mismatched": 0.518
|
| 60 |
+
},
|
| 61 |
+
"test_relation": {
|
| 62 |
+
"questions": 1950,
|
| 63 |
+
"accuracy": 0.7108,
|
| 64 |
+
"nll": 0.5401,
|
| 65 |
+
"ece": 0.0471,
|
| 66 |
+
"accuracy_mismatched": 0.4841
|
| 67 |
+
},
|
| 68 |
+
"test_vsr": {
|
| 69 |
+
"questions": 904,
|
| 70 |
+
"accuracy": 0.6792,
|
| 71 |
+
"nll": 0.6155,
|
| 72 |
+
"ece": 0.0478,
|
| 73 |
+
"accuracy_mismatched": 0.4812
|
| 74 |
+
},
|
| 75 |
+
"test_v7w": {
|
| 76 |
+
"questions": 1000,
|
| 77 |
+
"accuracy": 0.721,
|
| 78 |
+
"nll": 0.6838,
|
| 79 |
+
"ece": 0.0302,
|
| 80 |
+
"accuracy_mismatched": 0.422
|
| 81 |
+
},
|
| 82 |
+
"test_count_fresh": {
|
| 83 |
+
"questions": 600,
|
| 84 |
+
"accuracy": 0.74,
|
| 85 |
+
"nll": 0.5763,
|
| 86 |
+
"ece": 0.0648,
|
| 87 |
+
"accuracy_mismatched": 0.5067
|
| 88 |
+
}
|
| 89 |
+
},
|
| 90 |
+
"laya_vision": {
|
| 91 |
+
"vqav2_yesno": {
|
| 92 |
+
"questions": 4887,
|
| 93 |
+
"laya_vision": 0.7172,
|
| 94 |
+
"devision": 0.7248,
|
| 95 |
+
"difference": 0.0076,
|
| 96 |
+
"ci95": [
|
| 97 |
+
-0.0087,
|
| 98 |
+
0.0233
|
| 99 |
+
]
|
| 100 |
+
},
|
| 101 |
+
"aokvqa": {
|
| 102 |
+
"questions": 1138,
|
| 103 |
+
"laya_vision": 0.5984,
|
| 104 |
+
"devision": 0.6257,
|
| 105 |
+
"difference": 0.0272,
|
| 106 |
+
"ci95": [
|
| 107 |
+
-0.0062,
|
| 108 |
+
0.0598
|
| 109 |
+
]
|
| 110 |
+
},
|
| 111 |
+
"scienceqa": {
|
| 112 |
+
"questions": 2097,
|
| 113 |
+
"laya_vision": 0.824,
|
| 114 |
+
"devision": 0.7663,
|
| 115 |
+
"difference": -0.0577,
|
| 116 |
+
"ci95": [
|
| 117 |
+
-0.0792,
|
| 118 |
+
-0.0362
|
| 119 |
+
]
|
| 120 |
+
},
|
| 121 |
+
"scienceqa_natural_needs_picture": {
|
| 122 |
+
"questions": 323,
|
| 123 |
+
"laya_vision": 0.6997,
|
| 124 |
+
"devision": 0.4551,
|
| 125 |
+
"difference": -0.2446,
|
| 126 |
+
"ci95": [
|
| 127 |
+
-0.3158,
|
| 128 |
+
-0.1765
|
| 129 |
+
]
|
| 130 |
+
},
|
| 131 |
+
"pope": {
|
| 132 |
+
"random": {
|
| 133 |
+
"laya_vision_published": 0.836,
|
| 134 |
+
"devision": 0.8913
|
| 135 |
+
},
|
| 136 |
+
"popular": {
|
| 137 |
+
"laya_vision_published": 0.819,
|
| 138 |
+
"devision": 0.868
|
| 139 |
+
},
|
| 140 |
+
"adversarial": {
|
| 141 |
+
"laya_vision_published": 0.777,
|
| 142 |
+
"devision": 0.7913
|
| 143 |
+
}
|
| 144 |
+
}
|
| 145 |
+
},
|
| 146 |
+
"cpu_latency": {
|
| 147 |
+
"p50": 148,
|
| 148 |
+
"p95": 161,
|
| 149 |
+
"threads": 4,
|
| 150 |
+
"machine": "Darwin arm64",
|
| 151 |
+
"source": "runs/v12-ext/cpu/single.json"
|
| 152 |
+
}
|
| 153 |
+
}
|
evaluation/results.md
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# deVision v0.2 evaluation
|
| 2 |
+
|
| 3 |
+
Generated by `scripts/release/hf.py` from the evaluation of this release.
|
| 4 |
+
|
| 5 |
+
| Set | Questions | Accuracy | Mismatched | ECE |
|
| 6 |
+
|---|---:|---:|---:|---:|
|
| 7 |
+
| COCO object presence | 1,120 | 0.938 | 0.493 | 0.020 |
|
| 8 |
+
| VQAv2 multiple choice | 1,420 | 0.898 | 0.399 | 0.026 |
|
| 9 |
+
| COCO size | 1,100 | 0.860 | 0.504 | 0.036 |
|
| 10 |
+
| POPE, project filtered set | 8,676 | 0.851 | 0.527 | 0.058 |
|
| 11 |
+
| GQA val subset | 992 | 0.784 | 0.532 | 0.042 |
|
| 12 |
+
| COCO position | 1,274 | 0.872 | 0.493 | 0.025 |
|
| 13 |
+
| VQAv2 yes/no subset | 1,000 | 0.710 | 0.518 | 0.022 |
|
| 14 |
+
| COCO relative position | 1,950 | 0.711 | 0.484 | 0.047 |
|
| 15 |
+
| VSR, project held-out split | 904 | 0.679 | 0.481 | 0.048 |
|
| 16 |
+
| Visual7W, project held-out split | 1,000 | 0.721 | 0.422 | 0.030 |
|
| 17 |
+
| Fresh counting test (unseen pictures) | 600 | 0.740 | 0.507 | 0.065 |
|
| 18 |
+
|
| 19 |
+
## Laya Vision 201M, same questions
|
| 20 |
+
|
| 21 |
+
| Set | Questions | Laya Vision | deVision | Difference |
|
| 22 |
+
|---|---:|---:|---:|---|
|
| 23 |
+
| VQAv2 yes/no | 4,887 | 0.717 | 0.725 | +0.8 [-0.9, +2.3] |
|
| 24 |
+
| A-OKVQA | 1,138 | 0.598 | 0.626 | +2.7 [-0.6, +6.0] |
|
| 25 |
+
| ScienceQA with images | 2,097 | 0.824 | 0.766 | -5.8 [-7.9, -3.6] |
|
| 26 |
+
| ScienceQA natural science, needs the picture | 323 | 0.700 | 0.455 | -24.5 [-31.6, -17.6] |
|
| 27 |
+
|
| 28 |
+
On the full POPE random / popular / adversarial sets (3,000 questions each) deVision scores **0.891 / 0.868 / 0.791**, against Laya Vision's published **0.836 / 0.819 / 0.777** (aggregate scores only, not paired).
|
| 29 |
+
|
| 30 |
+
## Temperatures
|
| 31 |
+
|
| 32 |
+
| Bucket | Temperature |
|
| 33 |
+
|---|---:|
|
| 34 |
+
| Choice, 2 options | 1.8836 |
|
| 35 |
+
| Choice, 3–5 options | 1.9790 |
|
| 36 |
+
| Noul (yes / no) | 1.6382 |
|
| 37 |
+
| Choice, any other count | 1.9568 |
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:07a3e8aacb986396d3975d12d672526c6d3b8ec240b9d9d503ef235c5c87c3ae
|
| 3 |
+
size 2073754784
|
provenance.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"version": "v0.2",
|
| 3 |
+
"git_tag": "model-v0.2",
|
| 4 |
+
"built_from_commit": "bc1dedb9530935a7ed0eaebe61a9789dedb263bf",
|
| 5 |
+
"round": "v12-ext",
|
| 6 |
+
"stage": "stage2",
|
| 7 |
+
"training_commit": "43c21d692aa03594deb988934f8e88c39ffdb626+dirty",
|
| 8 |
+
"training_started": "2026-10-06 22:54:15",
|
| 9 |
+
"stages": [
|
| 10 |
+
{
|
| 11 |
+
"name": "stage2",
|
| 12 |
+
"out": "runs/v12-ext/stage2",
|
| 13 |
+
"init": "runs/v11-v9b-recovery/stage2"
|
| 14 |
+
}
|
| 15 |
+
],
|
| 16 |
+
"training_config": {
|
| 17 |
+
"name": "v12-ext",
|
| 18 |
+
"device": "mps",
|
| 19 |
+
"common": {
|
| 20 |
+
"log_every": 100
|
| 21 |
+
},
|
| 22 |
+
"stages": [
|
| 23 |
+
{
|
| 24 |
+
"name": "stage2",
|
| 25 |
+
"kind": "train",
|
| 26 |
+
"init": "runs/v11-v9b-recovery/stage2",
|
| 27 |
+
"data": "data/ext/train_mix.jsonl",
|
| 28 |
+
"val": "data/v5/dev_mix.jsonl",
|
| 29 |
+
"eval": {
|
| 30 |
+
"ext": "data/ext/dev_ext.jsonl"
|
| 31 |
+
},
|
| 32 |
+
"params": {
|
| 33 |
+
"epochs": 1,
|
| 34 |
+
"lr_new": 5e-05,
|
| 35 |
+
"lr_head": 5e-05,
|
| 36 |
+
"lr_lora": 0.0001,
|
| 37 |
+
"warmup": 500,
|
| 38 |
+
"eval_every": 5000,
|
| 39 |
+
"save_every": 1000,
|
| 40 |
+
"amp": "off",
|
| 41 |
+
"seed": 0
|
| 42 |
+
}
|
| 43 |
+
}
|
| 44 |
+
],
|
| 45 |
+
"evaluate": {
|
| 46 |
+
"image_identity": "data/v5/image_identity.json",
|
| 47 |
+
"sets": {
|
| 48 |
+
"dev_mix": "data/v5/dev_mix.jsonl",
|
| 49 |
+
"dev_lr_pairs": "data/v6/dev_lr_pairs.jsonl",
|
| 50 |
+
"dev_ext": {
|
| 51 |
+
"path": "data/ext/dev_ext.jsonl",
|
| 52 |
+
"role": "monitoring"
|
| 53 |
+
}
|
| 54 |
+
}
|
| 55 |
+
}
|
| 56 |
+
},
|
| 57 |
+
"source_checkpoint_sha256": "07a3e8aacb986396d3975d12d672526c6d3b8ec240b9d9d503ef235c5c87c3ae"
|
| 58 |
+
}
|
tokenizer/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
tokenizer/tokenizer_config.json
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"backend": "tokenizers",
|
| 3 |
+
"clean_up_tokenization_spaces": true,
|
| 4 |
+
"cls_token": "[CLS]",
|
| 5 |
+
"is_local": true,
|
| 6 |
+
"local_files_only": false,
|
| 7 |
+
"mask_token": "[MASK]",
|
| 8 |
+
"model_input_names": [
|
| 9 |
+
"input_ids",
|
| 10 |
+
"attention_mask"
|
| 11 |
+
],
|
| 12 |
+
"model_max_length": 8192,
|
| 13 |
+
"pad_token": "[PAD]",
|
| 14 |
+
"sep_token": "[SEP]",
|
| 15 |
+
"tokenizer_class": "TokenizersBackend",
|
| 16 |
+
"unk_token": "[UNK]"
|
| 17 |
+
}
|
vision/config.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"attention_dropout": 0.0,
|
| 3 |
+
"dtype": "float32",
|
| 4 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 5 |
+
"hidden_size": 768,
|
| 6 |
+
"image_size": 256,
|
| 7 |
+
"intermediate_size": 3072,
|
| 8 |
+
"layer_norm_eps": 1e-06,
|
| 9 |
+
"model_type": "siglip_vision_model",
|
| 10 |
+
"num_attention_heads": 12,
|
| 11 |
+
"num_channels": 3,
|
| 12 |
+
"num_hidden_layers": 12,
|
| 13 |
+
"patch_size": 16,
|
| 14 |
+
"transformers_version": "5.17.0"
|
| 15 |
+
}
|