demo: self-contained Gradio app (run locally)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +7 -0
- demo/LICENSE-APACHE +201 -0
- demo/LICENSE-MIT +21 -0
- demo/NOTICE +13 -0
- demo/app.py +516 -0
- demo/app_ru.json +23 -0
- demo/build_groups_v3.py +519 -0
- demo/configs/app.json +23 -0
- demo/configs/generator.json +58 -0
- demo/configs/speaking_rate.json +51 -0
- demo/configs/speaking_rate_ru.json +86 -0
- demo/examples_ru.json +1 -0
- demo/packages.txt +1 -0
- demo/requirements.txt +25 -0
- demo/runorm_cache/.locks/models--RUNorm--RUNorm-normalizer-medium/196088ae0adbafc572d99e62d0090a4a25a0bb305493ff454a7f9fb2171c1566.lock +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/added_tokens.json +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/chat_template.jinja +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/model.safetensors +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/model.safetensors.index.json +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer.model +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/1e39d930e09fd435aa8be1182640092b0581e3ca +5 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/23440d6773792a7758b5aab66878ba5f11f56f81235394a3856718c82e2a8f1e +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/49f05ecdcda8ba78b292f78ba2f83016d348cf62 +7 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/88ae2e52858aba88b4e935b2e5689ab4dc49bae6 +6 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/a5a347155014013f45c8adb3efaebf40d68142ed +29 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/c96c58140281ac3b8e2993004ebc727ff5b22171 +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/e33efa79e656bdd0972beb15a07b11b74df0df1368308bd12ec6b02337887040 +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/refs/main +1 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/refs/refs/pr/1 +1 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/config.json +29 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/generation_config.json +7 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/pytorch_model.bin +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/special_tokens_map.json +5 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer.json +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer_config.json +6 -0
- demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/c282bffa4b5a328a29883541fc4f8784f1ffe093/model.safetensors +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/chat_template.jinja +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/model.safetensors +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/model.safetensors.index.json +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/010ca0a172e0d662f661ff9b6f9b3323c1272ec4 +7 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/0e9c94dde3f01cda84a45287cabd0c22ab23cbac +61 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/12e7774c6c9be70934c0fd405e2fef3905609e14 +0 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/196088ae0adbafc572d99e62d0090a4a25a0bb305493ff454a7f9fb2171c1566.incomplete +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/4fe7c3c6498b6c26f0b5b6c60265b4d88ab4db02e1bbfb7f06cea3dc4746874b +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/5540ad7d1f6add43e0550aed386b556455404faa +108 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/5750775dfac6c13088a6c54dd07cf88f7f9cee1d +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/7a4eb87011448a4564a3144979384da51eee1da95e554feb22ccc85529535dd5 +3 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/94fe40201a0d25ffe6e448ae1a91f41fc67d28fa +938 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/refs/main +1 -0
- demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/refs/refs/pr/1 +1 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,10 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/23440d6773792a7758b5aab66878ba5f11f56f81235394a3856718c82e2a8f1e filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/e33efa79e656bdd0972beb15a07b11b74df0df1368308bd12ec6b02337887040 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/196088ae0adbafc572d99e62d0090a4a25a0bb305493ff454a7f9fb2171c1566.incomplete filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/4fe7c3c6498b6c26f0b5b6c60265b4d88ab4db02e1bbfb7f06cea3dc4746874b filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/7a4eb87011448a4564a3144979384da51eee1da95e554feb22ccc85529535dd5 filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
demo/runorm_cache/models--RUNorm--RUNorm-tagger/blobs/707128ea4c32fd7037ff0e49cd9b0234f0311f96e1b69ed413a7e90323b0b9e9 filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
demo/runorm_cache/models--RUNorm--RUNorm-tagger/blobs/e636f33b833a1e4e1c090ea2d7e0253288a86c4fff54c8b4d879ef7bb1b1ce0e filter=lfs diff=lfs merge=lfs -text
|
demo/LICENSE-APACHE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
demo/LICENSE-MIT
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2025 Nikita Torgashov
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
demo/NOTICE
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
NOTICE
|
| 2 |
+
|
| 3 |
+
This project includes the Depth Transformer component developed by SesameAI,
|
| 4 |
+
licensed under the Apache License, Version 2.0.
|
| 5 |
+
|
| 6 |
+
You may obtain a copy of the Apache License at:
|
| 7 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
|
| 9 |
+
Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
See the License for the specific language governing permissions and
|
| 13 |
+
limitations under the License.
|
demo/app.py
ADDED
|
@@ -0,0 +1,516 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""VoXtream2-RU — Gradio-демо (аналог HF space herimor/voxtream2, но с русской моделью).
|
| 2 |
+
|
| 3 |
+
Работает в двух режимах:
|
| 4 |
+
1) Локально: python ru_finetune/ft4/space/app.py --model-dir ru_finetune/ft4/infer_model
|
| 5 |
+
2) HF Space: переменная окружения MODEL_REPO=<user>/voxtream2-ru (repo с файлами
|
| 6 |
+
model.safetensors, config.json, phoneme_to_token.json, ru_tokens.json) —
|
| 7 |
+
файлы скачиваются через hf_hub_download.
|
| 8 |
+
|
| 9 |
+
Отличия от EN-демо: RUAccent ставит ударения (решает омографы за́мок/замо́к),
|
| 10 |
+
espeak-ru фонемизация, спец-токены расширенного словаря.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
import argparse
|
| 14 |
+
import json
|
| 15 |
+
import os
|
| 16 |
+
import re
|
| 17 |
+
import sys
|
| 18 |
+
from pathlib import Path
|
| 19 |
+
|
| 20 |
+
HERE = Path(__file__).resolve().parent
|
| 21 |
+
# HF Space: пакет voxtream и модули (v10_fix_dict, build_groups_v3) вендорены рядом
|
| 22 |
+
if str(HERE) not in sys.path:
|
| 23 |
+
sys.path.insert(0, str(HERE))
|
| 24 |
+
from v10_fix_dict import FIX as INTERJ_FIX, redup_phones as _interj_redup # noqa: E402
|
| 25 |
+
|
| 26 |
+
LOCAL_REPO_ID = "LOCAL_RU"
|
| 27 |
+
VOWEL = set("aeiouyɑɛɔʌəɵɨæøœɐɒʉʊɪ")
|
| 28 |
+
RU_VOWELS = set("аеёиоуыэюя")
|
| 29 |
+
PUNCT = (".", ",", "?", "!")
|
| 30 |
+
# v4: односимвольные согласные слова = проклитики (один звук), а не названия букв.
|
| 31 |
+
# espeak-ru даёт «к» -> k ˈɑ («ка») во ВСЕХ контекстах — источник жалобы
|
| 32 |
+
# «говорит "ка" вместо "к"». Совпадает с mfa_dict_v4.txt (обучение).
|
| 33 |
+
PROCLITIC = {"в": "v", "с": "s", "к": "k", "ж": "ʒ", "б": "b"}
|
| 34 |
+
STRIP_WORD = ".,?!—-«»\"'()…:;–„“”+"
|
| 35 |
+
# символы, которые НЕ являются фонемами: если прилипнут к фонеме, токен уходит в UNK
|
| 36 |
+
NON_PHONE = ".,?!—-«»\"'()…:;–„“”"
|
| 37 |
+
# после каких знаков вставлять паузу-токен (как в обучающих TextGrid)
|
| 38 |
+
SIL_AFTER = ".,!?"
|
| 39 |
+
|
| 40 |
+
# Модель знает только .,?! — остальные знаки нормализуем в ближайший по интонации
|
| 41 |
+
# (иначе espeak приклеивает их к фонеме: «добавил:» -> «ɭ:» -> UNK -> «добаве»).
|
| 42 |
+
_PUNCT_MAP = {":": ",", ";": ",", "—": ",", "–": ",", "…": ".", "«": "", "»": "",
|
| 43 |
+
"„": "", "“": "", "”": "", '"': "", "(": "", ")": ""}
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
TARGET_LUFS = -23.0 # = уровень корпуса v10 (клипы приводятся при рендере групп)
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def normalize_lufs(wav, sr):
|
| 50 |
+
"""v10: промпт -> -23 LUFS. Тихая бытовая запись для модели OOD; корпус
|
| 51 |
+
нормализован к тому же уровню (стандарт NeMo/NVIDIA voice cloning)."""
|
| 52 |
+
import numpy as np
|
| 53 |
+
import pyloudnorm
|
| 54 |
+
|
| 55 |
+
mono = wav.mean(dim=0).numpy().astype("float32")
|
| 56 |
+
if len(mono) < int(0.5 * sr): # метру BS.1770 нужно >= 0.4 c
|
| 57 |
+
return wav
|
| 58 |
+
try:
|
| 59 |
+
loud = pyloudnorm.Meter(sr).integrated_loudness(mono)
|
| 60 |
+
except Exception: # noqa: BLE001
|
| 61 |
+
return wav
|
| 62 |
+
if not np.isfinite(loud) or loud < -70:
|
| 63 |
+
return wav
|
| 64 |
+
out = wav * float(10 ** ((TARGET_LUFS - loud) / 20))
|
| 65 |
+
peak = float(out.abs().max())
|
| 66 |
+
if peak > 0.99: # true-peak защита
|
| 67 |
+
out = out * (0.99 / peak)
|
| 68 |
+
return out
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def condition_prompt(wav, sr):
|
| 72 |
+
"""v10.1: подготовка границы промпта. Замер: 5/6 тестовых промптов обрезаны
|
| 73 |
+
ПОСРЕДИ слова (жёсткий срез на 8 с), и провалы тембра сидят в первых ~1 с
|
| 74 |
+
генерации (окно 0.5 с: sim 0.11-0.29), а единственный промпт с тихим концом
|
| 75 |
+
(sanya_agin) — единственный без переключений. В обучении модель всегда
|
| 76 |
+
продолжает ПОСЛЕ паузы с комнатным тоном. Делаем так же:
|
| 77 |
+
1) если конец промпта — речь, режем назад до ближайшей тишины (>= 80 мс
|
| 78 |
+
ниже -35 dB от пика, в последних 2.5 с);
|
| 79 |
+
2) добавляем 0.35 с комнатного тона, синтезированного из тихого окна
|
| 80 |
+
самого промпта (build_groups_v3.synth_tone). VOXTREAM_PROMPT_COND=0 выкл."""
|
| 81 |
+
import numpy as np
|
| 82 |
+
if os.environ.get("VOXTREAM_PROMPT_COND", "1") != "1":
|
| 83 |
+
return wav
|
| 84 |
+
mono = wav.mean(dim=0).numpy().astype("float32")
|
| 85 |
+
n = len(mono)
|
| 86 |
+
fr = max(int(0.02 * sr), 1)
|
| 87 |
+
peak = float(np.abs(mono).max()) + 1e-9
|
| 88 |
+
if n > int(3.0 * sr):
|
| 89 |
+
# порог тишины адаптивный: 20-й перцентиль энергии 20-мс кадров записи
|
| 90 |
+
# (паузы ~20-30% речи), но не выше -25 дБ от пика (шумные бытовые записи)
|
| 91 |
+
nf_all = n // fr
|
| 92 |
+
e_all = 20 * np.log10(np.sqrt((mono[: nf_all * fr].reshape(nf_all, fr) ** 2).mean(1)) + 1e-9) - 20 * np.log10(peak)
|
| 93 |
+
thr = min(float(np.percentile(e_all, 20)), -25.0)
|
| 94 |
+
win = int(min(4.0 * sr, n - 2.0 * sr))
|
| 95 |
+
seg_e = e_all[-(win // fr):]
|
| 96 |
+
run, cut = 0, None
|
| 97 |
+
for i, q in enumerate(seg_e < thr):
|
| 98 |
+
run = run + 1 if q else 0
|
| 99 |
+
if run >= 3: # >= 60 мс тишины
|
| 100 |
+
cut = i
|
| 101 |
+
if cut is None:
|
| 102 |
+
# паузы нет (сплошная речь): режем в самом глубоком провале энергии
|
| 103 |
+
# последних 2 с (смычка/межсловный спад — не середина гласной)
|
| 104 |
+
tail_e = e_all[-int(1.2 * sr) // fr:]
|
| 105 |
+
deep = np.where(tail_e < float(np.median(e_all)) - 6.0)[0]
|
| 106 |
+
if len(deep): # ближайший к концу провал — теряем минимум промпта
|
| 107 |
+
cut = len(seg_e) - len(tail_e) + int(deep[-1]) + 1
|
| 108 |
+
if cut is not None:
|
| 109 |
+
end = (nf_all - len(seg_e) + cut - 1) * fr # внутри найденной тишины/провала
|
| 110 |
+
if end > int(2.0 * sr):
|
| 111 |
+
wav = wav[:, :end]
|
| 112 |
+
mono = mono[:end]
|
| 113 |
+
try:
|
| 114 |
+
import build_groups_v3 as B
|
| 115 |
+
tmpl = B.room_tone(mono)
|
| 116 |
+
tone = B.synth_tone(tmpl, int(0.35 * sr), seed=7).astype("float32")
|
| 117 |
+
except Exception: # noqa: BLE001
|
| 118 |
+
tone = (np.random.randn(int(0.35 * sr)) * 1e-4).astype("float32")
|
| 119 |
+
import torch
|
| 120 |
+
tone_t = torch.from_numpy(tone)[None].expand(wav.shape[0], -1)
|
| 121 |
+
return torch.cat([wav, tone_t.to(wav.dtype)], dim=1)
|
| 122 |
+
|
| 123 |
+
|
| 124 |
+
def normalize_punct(text: str) -> str:
|
| 125 |
+
for src, dst in _PUNCT_MAP.items():
|
| 126 |
+
text = text.replace(src, dst)
|
| 127 |
+
text = re.sub(r"\s+([,.!?])", r"\1", text) # пробел перед знаком
|
| 128 |
+
text = re.sub(r"([,.!?])\1+", r"\1", text) # дубли знаков
|
| 129 |
+
return re.sub(r"\s{2,}", " ", text).strip()
|
| 130 |
+
|
| 131 |
+
|
| 132 |
+
def is_vowel(t):
|
| 133 |
+
return any(c in VOWEL for c in t)
|
| 134 |
+
|
| 135 |
+
|
| 136 |
+
def stressed_idx(aw):
|
| 137 |
+
vi, cnt, s, i = -1, 0, aw.lower(), 0
|
| 138 |
+
while i < len(s):
|
| 139 |
+
if s[i] == "+":
|
| 140 |
+
if i + 1 < len(s) and s[i + 1] in RU_VOWELS:
|
| 141 |
+
vi = cnt
|
| 142 |
+
i += 1
|
| 143 |
+
continue
|
| 144 |
+
if s[i] in RU_VOWELS:
|
| 145 |
+
cnt += 1
|
| 146 |
+
i += 1
|
| 147 |
+
return vi
|
| 148 |
+
|
| 149 |
+
|
| 150 |
+
class RUAccentPhonemizer:
|
| 151 |
+
"""Интерфейс ESpeak.phonemize + RUAccent-ударения (перенос ˈ на нужную гласную)."""
|
| 152 |
+
|
| 153 |
+
def __init__(self):
|
| 154 |
+
from ruaccent import RUAccent
|
| 155 |
+
from voxtream.utils.text.phonemizer import ESpeak
|
| 156 |
+
self.acc = RUAccent()
|
| 157 |
+
self.acc.load(omograph_model_size="turbo3.1", use_dictionary=True)
|
| 158 |
+
self.esp = ESpeak("ru")
|
| 159 |
+
self._runorm = None # ленивая загрузка: нужна только для текстов с цифрами
|
| 160 |
+
|
| 161 |
+
def _normalize_digits(self, text: str) -> str:
|
| 162 |
+
"""v10: цифры/числа -> слова (RUNorm, падежи/род учитывает). Пользователь
|
| 163 |
+
пишет «в 2024 году» — работает без ручной нормализации. «Ё», потерянную
|
| 164 |
+
RUNorm'ом («четвертом»), ниже восстановит RUAccent."""
|
| 165 |
+
if not re.search(r"\d", text):
|
| 166 |
+
return text
|
| 167 |
+
if self._runorm is None:
|
| 168 |
+
from runorm import RUNorm
|
| 169 |
+
self._runorm = RUNorm()
|
| 170 |
+
self._runorm.load(model_size="medium",
|
| 171 |
+
workdir=str(HERE / "runorm_cache"))
|
| 172 |
+
# только предложения С цифрами: на остальных RUNorm вредит — разворачивает
|
| 173 |
+
# междометия как аббревиатуры («Ммм» -> «эм эм эм») мимо нашей таблицы
|
| 174 |
+
try:
|
| 175 |
+
parts = re.split(r"(?<=[.!?…])\s+", text)
|
| 176 |
+
return " ".join(
|
| 177 |
+
self._runorm.norm(p) if re.search(r"\d", p) else p for p in parts
|
| 178 |
+
)
|
| 179 |
+
except Exception as e: # noqa: BLE001 — цифры хуже, чем необработанный текст
|
| 180 |
+
print(f"[runorm] fail: {e}; текст без нормализации")
|
| 181 |
+
return text
|
| 182 |
+
|
| 183 |
+
def _accent(self, text: str) -> str:
|
| 184 |
+
"""Ударения: ручной '+' перед гласной (зам+ок) имеет приоритет,
|
| 185 |
+
RUAccent ставит только в словах без ручной пометки."""
|
| 186 |
+
if "+" not in text:
|
| 187 |
+
return self.acc.process_all(text)
|
| 188 |
+
plain = re.sub(r"\+", "", text)
|
| 189 |
+
auto = self.acc.process_all(plain)
|
| 190 |
+
manual_w, auto_w = text.split(), auto.split()
|
| 191 |
+
if len(manual_w) != len(auto_w):
|
| 192 |
+
return auto # рассинхрон токенизации — безопасный фолбэк
|
| 193 |
+
return " ".join(
|
| 194 |
+
mw if "+" in mw else aw for mw, aw in zip(manual_w, auto_w)
|
| 195 |
+
)
|
| 196 |
+
|
| 197 |
+
def phonemize(self, text, separator="|", language="ru"):
|
| 198 |
+
text = self._normalize_digits(text)
|
| 199 |
+
text = normalize_punct(text)
|
| 200 |
+
accented = self._accent(text)
|
| 201 |
+
clean = re.sub(r"\+", "", accented)
|
| 202 |
+
seq = self.esp.phonemize(clean, separator=separator, language="ru")
|
| 203 |
+
esp_words, acc_words = seq.split(), accented.split()
|
| 204 |
+
if len(esp_words) != len(acc_words):
|
| 205 |
+
return seq
|
| 206 |
+
out = []
|
| 207 |
+
for ew, aw in zip(esp_words, acc_words):
|
| 208 |
+
phones = [p for p in ew.split(separator) if p]
|
| 209 |
+
# фикс v4: espeak озвучивает предлог «к» как НАЗВАНИЕ буквы (k ˈɑ ->
|
| 210 |
+
# «ка тебе»). Односимвольные согласные слова — проклитики в один звук.
|
| 211 |
+
bare = aw.strip(STRIP_WORD).lower()
|
| 212 |
+
if bare in PROCLITIC:
|
| 213 |
+
keep = PROCLITIC[bare]
|
| 214 |
+
tail_p = phones[-1][-1] if phones and phones[-1][-1] in "".join(PUNCT) else ""
|
| 215 |
+
phones = [keep + tail_p] if tail_p else [keep]
|
| 216 |
+
# v10: междометия (хм/ммм/тсс...) espeak читает НАЗВАНИЯМИ букв
|
| 217 |
+
# («ха-эм»); в словаре v5 они — чистые согласные. Таблица общая
|
| 218 |
+
# с v10_fix_dict, чтобы цикл обучение<->инференс не расходился.
|
| 219 |
+
elif (ij := INTERJ_FIX.get(bare) or _interj_redup(bare)) is not None:
|
| 220 |
+
tail_p = phones[-1][-1] if phones and phones[-1][-1] in "".join(PUNCT) else ""
|
| 221 |
+
phones = ij.split()
|
| 222 |
+
if tail_p:
|
| 223 |
+
phones[-1] += tail_p
|
| 224 |
+
tail = ""
|
| 225 |
+
if phones and phones[-1] and phones[-1][-1] in "".join(PUNCT):
|
| 226 |
+
tail = phones[-1][-1]
|
| 227 |
+
phones[-1] = phones[-1][:-1]
|
| 228 |
+
if not phones[-1]:
|
| 229 |
+
phones.pop()
|
| 230 |
+
# страховка: любой НЕ-фонемный хвост (":", ";", "»", ")"…) прилипает к
|
| 231 |
+
# последней фонеме -> токен «ɭ:» отсутствует в словаре -> UNK -> звук
|
| 232 |
+
# пропадает («добавил:» звучало как «добаве»). Чистим остатки.
|
| 233 |
+
phones = [p for p in (q.strip(NON_PHONE) for q in phones) if p]
|
| 234 |
+
ti = stressed_idx(aw)
|
| 235 |
+
if ti >= 0 and phones:
|
| 236 |
+
cl = [p.replace("ˈ", "").replace("ˌ", "") for p in phones]
|
| 237 |
+
vp = [j for j, p in enumerate(cl) if is_vowel(p)]
|
| 238 |
+
if ti < len(vp):
|
| 239 |
+
cl[vp[ti]] = "ˈ" + cl[vp[ti]]
|
| 240 |
+
phones = cl
|
| 241 |
+
w = separator.join(phones)
|
| 242 |
+
out.append(w + tail if tail else w)
|
| 243 |
+
# ФИКС ПАУЗ: в обучении паузы стоят в потоке ЯВНЫМИ токенами 'sil'
|
| 244 |
+
# (MFA-разметка, 6.3% токенов), а espeak их не даёт — модель получала
|
| 245 |
+
# непрерывный поток и произносила всё слитно (на «Раз, два, три,
|
| 246 |
+
# четыре, пять.» — НОЛЬ пауз; звуки проглатывались). Вставляем sil
|
| 247 |
+
# после знака: длительность паузы модель выберет сама.
|
| 248 |
+
if tail and tail in SIL_AFTER:
|
| 249 |
+
out.append("sil")
|
| 250 |
+
return " ".join(out)
|
| 251 |
+
|
| 252 |
+
|
| 253 |
+
def resolve_model_files():
|
| 254 |
+
"""-> dict имя_файла -> локальный путь (из --model-dir или MODEL_REPO)."""
|
| 255 |
+
ap = argparse.ArgumentParser()
|
| 256 |
+
ap.add_argument("--model-dir", default=os.environ.get("MODEL_DIR", ""))
|
| 257 |
+
args, _ = ap.parse_known_args()
|
| 258 |
+
|
| 259 |
+
names = ["model.safetensors", "config.json", "phoneme_to_token.json", "ru_tokens.json"]
|
| 260 |
+
if args.model_dir:
|
| 261 |
+
d = Path(args.model_dir).resolve()
|
| 262 |
+
return {n: str(d / n) for n in names}
|
| 263 |
+
repo = os.environ.get("MODEL_REPO")
|
| 264 |
+
assert repo, "укажите --model-dir или env MODEL_REPO"
|
| 265 |
+
from huggingface_hub import hf_hub_download
|
| 266 |
+
return {n: hf_hub_download(repo, n) for n in names}
|
| 267 |
+
|
| 268 |
+
|
| 269 |
+
def main():
|
| 270 |
+
files = resolve_model_files()
|
| 271 |
+
ru_tokens = json.load(open(files["ru_tokens.json"]))
|
| 272 |
+
|
| 273 |
+
# --- монки-патчи ДО импорта app ---
|
| 274 |
+
import voxtream.utils.generator.setup as S
|
| 275 |
+
import voxtream.utils.generator.text as T
|
| 276 |
+
from huggingface_hub import hf_hub_download as _real_hf
|
| 277 |
+
|
| 278 |
+
def _hf(repo_id, filename, **kw):
|
| 279 |
+
if repo_id == LOCAL_REPO_ID:
|
| 280 |
+
return files[filename]
|
| 281 |
+
return _real_hf(repo_id, filename, **kw)
|
| 282 |
+
|
| 283 |
+
S.hf_hub_download = _hf
|
| 284 |
+
|
| 285 |
+
_orig_ttp = T.text_to_phone_tokens
|
| 286 |
+
|
| 287 |
+
# E1 (question-prefix модели): '?' после первого слова вопросительного
|
| 288 |
+
# предложения — как в обучении (PT видит «впереди вопрос» с самого начала).
|
| 289 |
+
# Включается флагом QUESTION_PREFIX=1 (для infer_model_e).
|
| 290 |
+
q_prefix = os.environ.get("QUESTION_PREFIX", "0") == "1"
|
| 291 |
+
|
| 292 |
+
def _add_q_prefix(text: str) -> str:
|
| 293 |
+
out = []
|
| 294 |
+
for sent in re.split(r"(?<=[.!?])\s+", str(text).strip()):
|
| 295 |
+
words = sent.split()
|
| 296 |
+
if sent.rstrip().endswith("?") and len(words) > 1:
|
| 297 |
+
words[0] += "?"
|
| 298 |
+
out.append(" ".join(words))
|
| 299 |
+
return " ".join(out)
|
| 300 |
+
|
| 301 |
+
def _ttp(*a, **kw):
|
| 302 |
+
kw["normalize"] = False
|
| 303 |
+
kw["language"] = "ru"
|
| 304 |
+
if q_prefix and a and isinstance(a[0], str):
|
| 305 |
+
a = (_add_q_prefix(a[0]),) + a[1:]
|
| 306 |
+
elif q_prefix and "text" in kw:
|
| 307 |
+
kw["text"] = _add_q_prefix(kw["text"])
|
| 308 |
+
return _orig_ttp(*a, **kw)
|
| 309 |
+
|
| 310 |
+
T.text_to_phone_tokens = _ttp
|
| 311 |
+
|
| 312 |
+
# v10: LUFS-нормализация промпта — шим над torchaudio ТОЛЬКО внутри prompt.py
|
| 313 |
+
# (глобальный torchaudio.load не трогаем). Внимание: .prompt.npy-кэши,
|
| 314 |
+
# созданные до нормализации, устаревают — их надо удалить.
|
| 315 |
+
import voxtream.utils.generator.prompt as PR
|
| 316 |
+
_orig_ta = PR.torchaudio
|
| 317 |
+
|
| 318 |
+
class _LufsTorchaudio:
|
| 319 |
+
def __getattr__(self, name):
|
| 320 |
+
return getattr(_orig_ta, name)
|
| 321 |
+
|
| 322 |
+
@staticmethod
|
| 323 |
+
def load(path, *a, **kw):
|
| 324 |
+
wav, sr = _orig_ta.load(path, *a, **kw)
|
| 325 |
+
return condition_prompt(normalize_lufs(wav, sr), sr), sr
|
| 326 |
+
|
| 327 |
+
PR.torchaudio = _LufsTorchaudio()
|
| 328 |
+
|
| 329 |
+
from voxtream.generator import SpeechGenerator
|
| 330 |
+
_orig_init = SpeechGenerator.__init__
|
| 331 |
+
|
| 332 |
+
def _patched_init(self, *a, **kw):
|
| 333 |
+
_orig_init(self, *a, **kw)
|
| 334 |
+
self.ctx.phonemizer = RUAccentPhonemizer()
|
| 335 |
+
|
| 336 |
+
SpeechGenerator.__init__ = _patched_init
|
| 337 |
+
|
| 338 |
+
# v10.1: синтез ПО ПРЕДЛОЖЕНИЯМ с повторной привязкой к промпту.
|
| 339 |
+
# Замер 384 сэмплов: переключение голоса после паузы между предложениями —
|
| 340 |
+
# 12-16% на многопредложенческих фразах против 3% на одиночных; трудные
|
| 341 |
+
# голоса (shibakov 33%) теряются после паузы. Каждое предложение стартует
|
| 342 |
+
# сразу после промпта, где привязка максимальна; между ними — пауза
|
| 343 |
+
# (медиана корпуса по знаку). VOXTREAM_SENT_SPLIT=0 выключает.
|
| 344 |
+
_orig_gs = SpeechGenerator.generate_stream
|
| 345 |
+
_PAUSE = {".": 0.39, "?": 0.42, "!": 0.41} # эмпирика 4.5 млн пауз (build_groups)
|
| 346 |
+
|
| 347 |
+
def _split_sentences(text: str):
|
| 348 |
+
parts = [p.strip() for p in re.split(r"(?<=[.!?…])\s+", text.strip()) if p.strip()]
|
| 349 |
+
return parts if len(parts) >= 2 else [text]
|
| 350 |
+
|
| 351 |
+
def _gs_split(self, prompt_audio_path, text, speaking_rate=None, enhance_prompt=None,
|
| 352 |
+
apply_vad=None, return_progress=False, min_streaming_rtf=None):
|
| 353 |
+
if os.environ.get("VOXTREAM_SENT_SPLIT", "1") != "1" or not isinstance(text, str):
|
| 354 |
+
yield from _orig_gs(self, prompt_audio_path, text, speaking_rate, enhance_prompt,
|
| 355 |
+
apply_vad, return_progress, min_streaming_rtf)
|
| 356 |
+
return
|
| 357 |
+
sents = _split_sentences(text)
|
| 358 |
+
if len(sents) < 2:
|
| 359 |
+
yield from _orig_gs(self, prompt_audio_path, text, speaking_rate, enhance_prompt,
|
| 360 |
+
apply_vad, return_progress, min_streaming_rtf)
|
| 361 |
+
return
|
| 362 |
+
import numpy as _np
|
| 363 |
+
sr = int(self.config.mimi_sr)
|
| 364 |
+
pos_off, time_off, last_prog = 0, 0.0, None
|
| 365 |
+
for k, sent in enumerate(sents):
|
| 366 |
+
last_pos, last_t = 0, 0.0
|
| 367 |
+
for item in _orig_gs(self, prompt_audio_path, sent, speaking_rate, enhance_prompt,
|
| 368 |
+
apply_vad, return_progress, min_streaming_rtf):
|
| 369 |
+
if return_progress:
|
| 370 |
+
frame, gt, prog = item
|
| 371 |
+
prog = dict(prog)
|
| 372 |
+
last_pos = max(last_pos, int(prog.get("phone_position", 0) or 0))
|
| 373 |
+
last_t = max(last_t, float(prog.get("time_sec", 0.0) or 0.0))
|
| 374 |
+
prog["phone_position"] = pos_off + int(prog.get("phone_position", 0) or 0)
|
| 375 |
+
prog["time_sec"] = time_off + float(prog.get("time_sec", 0.0) or 0.0)
|
| 376 |
+
last_prog = prog
|
| 377 |
+
yield frame, gt, prog
|
| 378 |
+
else:
|
| 379 |
+
yield item
|
| 380 |
+
pos_off += last_pos + 1
|
| 381 |
+
time_off += last_t
|
| 382 |
+
if k < len(sents) - 1:
|
| 383 |
+
gap = _PAUSE.get(sent[-1], 0.39)
|
| 384 |
+
n = int(gap * sr)
|
| 385 |
+
noise = (_np.random.randn(n) * 1e-4).astype(_np.float32) # не «цифровой» ноль
|
| 386 |
+
time_off += gap
|
| 387 |
+
if return_progress:
|
| 388 |
+
prog = dict(last_prog or {})
|
| 389 |
+
prog["time_sec"] = time_off
|
| 390 |
+
yield noise, 0.0, prog
|
| 391 |
+
else:
|
| 392 |
+
yield noise, 0.0
|
| 393 |
+
|
| 394 |
+
# v10.1: BEST-OF-N по удержанию голоса (offline-режим). Замер: 8-12% сэмплов
|
| 395 |
+
# с провалом оконной spk-sim к промпту (>0.25 от медианы), реальная речь тех
|
| 396 |
+
# же людей — 0/8; ни данные (S2b), ни сэмплирование/CFG/SRC/подготовка
|
| 397 |
+
# промпта цифру не сдвинули. Best-of-2 -> ~0.7% ожидаемо. Скорим тем же
|
| 398 |
+
# ReDimNet, что кондиционирует модель (ctx.spk_enc). VOXTREAM_BEST_OF=1 выкл.
|
| 399 |
+
from voxtream.utils.generator.prompt import extract_speaker_template as _est
|
| 400 |
+
import voxtream.utils.generator.prompt as _PR
|
| 401 |
+
|
| 402 |
+
def _spk_score(self, prompt_audio_path, audio_np):
|
| 403 |
+
import numpy as _np
|
| 404 |
+
import torch as _t
|
| 405 |
+
wav, sr = _PR.torchaudio.load(str(prompt_audio_path)) # шим: LUFS + граница
|
| 406 |
+
pe = _est(wav.mean(0, keepdim=True), sr, self.ctx.spk_enc, self.config.spk_enc_sr,
|
| 407 |
+
self.ctx.device, self.ctx.dtype).float().reshape(-1)
|
| 408 |
+
gen = _t.from_numpy(audio_np.astype("float32"))[None]
|
| 409 |
+
osr = int(self.config.mimi_sr)
|
| 410 |
+
W, H = int(1.5 * osr), int(0.5 * osr)
|
| 411 |
+
sims = []
|
| 412 |
+
for st in range(0, max(gen.shape[1] - W, 1), H):
|
| 413 |
+
e = _est(gen[:, st:st + W], osr, self.ctx.spk_enc, self.config.spk_enc_sr,
|
| 414 |
+
self.ctx.device, self.ctx.dtype).float().reshape(-1)
|
| 415 |
+
sims.append(float(pe @ e))
|
| 416 |
+
if not sims:
|
| 417 |
+
return 0.0, 0.0
|
| 418 |
+
return float(min(sims)), float(_np.median(sims))
|
| 419 |
+
|
| 420 |
+
def _gs_bestof(self, prompt_audio_path, text, speaking_rate=None, enhance_prompt=None,
|
| 421 |
+
apply_vad=None, return_progress=False, min_streaming_rtf=None):
|
| 422 |
+
try:
|
| 423 |
+
N = int(os.environ.get("VOXTREAM_BEST_OF", "2"))
|
| 424 |
+
except ValueError:
|
| 425 |
+
N = 1
|
| 426 |
+
if N <= 1 or not isinstance(text, str):
|
| 427 |
+
yield from _gs_split(self, prompt_audio_path, text, speaking_rate, enhance_prompt,
|
| 428 |
+
apply_vad, return_progress, min_streaming_rtf)
|
| 429 |
+
return
|
| 430 |
+
import numpy as _np
|
| 431 |
+
best, best_key = None, None
|
| 432 |
+
for k in range(N):
|
| 433 |
+
items = list(_gs_split(self, prompt_audio_path, text, speaking_rate, enhance_prompt,
|
| 434 |
+
apply_vad, return_progress, min_streaming_rtf))
|
| 435 |
+
audio = _np.concatenate([it[0] for it in items]) if items else _np.zeros(1, "float32")
|
| 436 |
+
mn, med = _spk_score(self, prompt_audio_path, audio)
|
| 437 |
+
key = (mn >= med - 0.25, mn) # сначала «без провала», затем по худшему окну
|
| 438 |
+
print(f"[best-of-{N}] кандидат {k + 1}: sim_min {mn:.2f} med {med:.2f}", flush=True)
|
| 439 |
+
if best_key is None or key > best_key:
|
| 440 |
+
best, best_key = items, key
|
| 441 |
+
if best_key[0]: # чистый кандидат — хватит (перегенерация только при провале)
|
| 442 |
+
break
|
| 443 |
+
yield from best
|
| 444 |
+
|
| 445 |
+
SpeechGenerator.generate_stream = _gs_bestof
|
| 446 |
+
|
| 447 |
+
# --- конфиги: базовые из репо + RU-переопределения ---
|
| 448 |
+
root = HERE
|
| 449 |
+
gen_cfg = json.load(open(root / "configs/generator.json"))
|
| 450 |
+
gen_cfg.update(
|
| 451 |
+
model_repo=LOCAL_REPO_ID,
|
| 452 |
+
unk_token=ru_tokens["unk"], eop_token=ru_tokens["unk"],
|
| 453 |
+
bos_token=ru_tokens["bos"], eos_token=ru_tokens["eos"],
|
| 454 |
+
sil_token=ru_tokens["sil"],
|
| 455 |
+
enhance_prompt=False, apply_vad=False,
|
| 456 |
+
# v10: свип 12 ячеек temp x topk на промптах пользователя (v9, 2026-08-08):
|
| 457 |
+
# 0.8/50 -> CER 0.0105, CV 0.159 против 0.023/0.177 у унаследованных 0.9/5;
|
| 458 |
+
# t1.0/k5 давал CER 0.0096, но худший CV 0.202 и темп 4.89 — не взят.
|
| 459 |
+
temperature=0.8, topk=50,
|
| 460 |
+
# v4 (свипы на ПРОМПТАХ ПОЛЬЗОВАТЕЛЯ ru_finetune/unseen — «удобные»
|
| 461 |
+
# корпусные промпты регрессию не ловили): при gain=25 cfg_gamma 1.5 даёт
|
| 462 |
+
# CER 0.021 против 0.056 у 2.0 и 0.024 у 3.0; unseen spk_sim 0.602 против
|
| 463 |
+
# 0.616 у 3.0, НО худший случай 0.422 против 0.269 — провалов нет.
|
| 464 |
+
cfg_gamma=1.5,
|
| 465 |
+
)
|
| 466 |
+
# Усиление SPS-кондиционирования. gain=40 подбирался на stage D, где бины
|
| 467 |
+
# считались по БУКВАМ и эмбеддинг почти не обучался. В v4 бины из ФОНЕМ —
|
| 468 |
+
# эмбеддинг информативнее, и 40 РАЗРУШАЕТ генерацию: на промптах пользователя
|
| 469 |
+
# CER 0.366 и скачки темпа (жалоба «неразборчиво, то быстро то медленно»).
|
| 470 |
+
# Свип по CER: gain 1-20 → 0.016-0.019, 25 → 0.024, 30 → 0.075, 40 → 0.366.
|
| 471 |
+
# 25 — компромисс: разборчивость вдвое лучше v3b при самом медленном темпе
|
| 472 |
+
# среди «чистых» режимов. Замедление ниже ~5.7 SPS этим механизмом ЛОМАЕТ речь
|
| 473 |
+
# (gain20/rate3.5 → CER 0.121; gain30/rate3.5 → 0.584) — темп чинить обучением.
|
| 474 |
+
os.environ.setdefault("VOXTREAM_SPS_GAIN", "25")
|
| 475 |
+
# Управление темпом: SRC вместо SPS-эмбеддинга. Замер на v7 (промпты юзера):
|
| 476 |
+
# SPS gain30/rate3.5 -> темп 4.64 но CER 0.245 (речь рвётся); SRC rate3.5 ->
|
| 477 |
+
# темп 5.37 при CER 0.025 и паузах 0.39с (= корпусная норма). SRC не искажает
|
| 478 |
+
# эмбеддинг, поэтому замедление безопасно.
|
| 479 |
+
os.environ.setdefault("VOXTREAM_TEMPO_MODE", "src")
|
| 480 |
+
# v10.1: лимит удержания фонемы по классу (кадры по 80 мс). Жалоба «буквы
|
| 481 |
+
# повторяются» = залипание на последней согласной фразы («Потомммм,» — 26
|
| 482 |
+
# кадров при защите frame_repeat_counter=25). В данных 97% согласных перед
|
| 483 |
+
# знаком <= 5 кадров, 99% гласных <= 12. A/B на S2-e2: без лимита макс согл.
|
| 484 |
+
# 11 кадров, CER 0.029; с лимитом — макс 5-6, CER 0.017-0.023, тембр не хуже.
|
| 485 |
+
os.environ.setdefault("VOXTREAM_DWELL_CAPS", "cons=5,vow=12")
|
| 486 |
+
|
| 487 |
+
ru_gen = HERE / "generator_ru.json"
|
| 488 |
+
# точечные переопределения инференс-параметров: GEN_OVERRIDES='{"cfg_gamma":2.0}'
|
| 489 |
+
overrides = os.environ.get("GEN_OVERRIDES")
|
| 490 |
+
if overrides:
|
| 491 |
+
gen_cfg.update(json.loads(overrides))
|
| 492 |
+
print(f"[app] generator overrides: {overrides}")
|
| 493 |
+
json.dump(gen_cfg, open(ru_gen, "w"), indent=2)
|
| 494 |
+
|
| 495 |
+
examples = HERE / "examples_ru.json"
|
| 496 |
+
if not examples.exists():
|
| 497 |
+
json.dump({"examples": []}, open(examples, "w"))
|
| 498 |
+
|
| 499 |
+
sys.argv = [
|
| 500 |
+
"voxtream-app",
|
| 501 |
+
"--config", str(ru_gen),
|
| 502 |
+
"--app-config", str(p if (p := HERE / "app_ru.json").exists()
|
| 503 |
+
else root / "configs/app.json"),
|
| 504 |
+
# RU-гистограммы SRC (пересчитаны по ft4-корпусу), фолбэк — EN-оригинал
|
| 505 |
+
"--spk-rate-config", str(
|
| 506 |
+
p if (p := root / "configs/speaking_rate_ru.json").exists()
|
| 507 |
+
else root / "configs/speaking_rate.json"
|
| 508 |
+
),
|
| 509 |
+
"--examples-config", str(examples),
|
| 510 |
+
]
|
| 511 |
+
from voxtream.app import main as app_main
|
| 512 |
+
app_main()
|
| 513 |
+
|
| 514 |
+
|
| 515 |
+
if __name__ == "__main__":
|
| 516 |
+
main()
|
demo/app_ru.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"min_chunk_sec": 0.01,
|
| 3 |
+
"fade_out_sec": 0.1,
|
| 4 |
+
"plot_window_sec": 10.0,
|
| 5 |
+
"visual_update_sec": 0.25,
|
| 6 |
+
"future_phone_limit": 25,
|
| 7 |
+
"plot_width": 1000,
|
| 8 |
+
"plot_height": 224,
|
| 9 |
+
"plot_left": 74,
|
| 10 |
+
"plot_right": 22,
|
| 11 |
+
"plot_top": 50,
|
| 12 |
+
"plot_bottom": 42,
|
| 13 |
+
"plot_y_max": 7,
|
| 14 |
+
"plot_y_tick": 1,
|
| 15 |
+
"plot_x_tick_sec": 1,
|
| 16 |
+
"audio_stream_start_delay_sec": 0.12,
|
| 17 |
+
"audio_stream_sample_rate": 24000,
|
| 18 |
+
"speaking_rate_min": 1.0,
|
| 19 |
+
"speaking_rate_max": 7.0,
|
| 20 |
+
"speaking_rate_step": 0.1,
|
| 21 |
+
"speaking_rate_default": 3.5,
|
| 22 |
+
"min_streaming_rtf": 0.95
|
| 23 |
+
}
|
demo/build_groups_v3.py
ADDED
|
@@ -0,0 +1,519 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""v3: пересборка 55-с групп с ЕСТЕСТВЕННЫМИ стыками (лечит «обрыв на точке»,
|
| 2 |
+
константные паузы и мёртвые нули между клипами).
|
| 3 |
+
|
| 4 |
+
Отличия от build_groups_ft4 (v2):
|
| 5 |
+
1. Паузы на стыках — семплируются из ЭМПИРИЧЕСКИХ распределений реальных пауз
|
| 6 |
+
корпуса (v3/pause_hist.json, посчитан по 1.14M по-клиповых MFA-выравниваний),
|
| 7 |
+
класс — по хвостовому знаку клипа: медианы ~0.39s после '.', 0.42s после '?',
|
| 8 |
+
0.27s после ',' (в v2 были константы 160/100 мс — в 2.5 раза короче естественных).
|
| 9 |
+
2. Фейды адаптивные: до 40 мс приподнятым косинусом, но ТОЛЬКО внутри собственной
|
| 10 |
+
краевой тишины клипа (речь не трогаем; минимум 2 мс от щелчков).
|
| 11 |
+
3. Паузы и хвост группы — не цифровой ноль, а КОМНАТНЫЙ ТОН клипа (самое тихое
|
| 12 |
+
60-мс окно, тайлится зеркально — C0-непрерывно, с джиттером амплитуды).
|
| 13 |
+
4. Клипы без MFA-выравнивания отсекаются на этапе плана (в v2 группы с ними
|
| 14 |
+
молча выпадали на этапе npy целиком).
|
| 15 |
+
5. Оверсемпл вопросов (x2 прохода) и короткие группы (10%, 5-35 c) — встроены
|
| 16 |
+
(в v2 это делал отдельный extend_groups_e).
|
| 17 |
+
|
| 18 |
+
Порядок клипов внутри спикера = порядок манифеста (для аудиокниг/подкастов это
|
| 19 |
+
порядок записи -> соседние клипы часто из одной сессии).
|
| 20 |
+
|
| 21 |
+
Выход: ft4/groups_v3.json, wav в voxtream_ru/_groups55_v3/, grouped_v3_chunk*.parquet.
|
| 22 |
+
|
| 23 |
+
Запуск:
|
| 24 |
+
.venv/bin/python ru/pipeline/build_groups_v3.py --plan-only # только план
|
| 25 |
+
.venv/bin/python ru/pipeline/build_groups_v3.py --workers 24 # план + wav
|
| 26 |
+
"""
|
| 27 |
+
|
| 28 |
+
import argparse
|
| 29 |
+
import json
|
| 30 |
+
import multiprocessing as mp
|
| 31 |
+
import random
|
| 32 |
+
import re
|
| 33 |
+
from pathlib import Path
|
| 34 |
+
|
| 35 |
+
import numpy as np
|
| 36 |
+
import pandas as pd
|
| 37 |
+
import soundfile as sf
|
| 38 |
+
|
| 39 |
+
ROOT = Path("/mnt/data/voxtream/ru_finetune/ru/data")
|
| 40 |
+
MFA_OUT = ROOT / "mfa_out"
|
| 41 |
+
GROUP_DIR = Path("/mnt/data/audio_data/voxtream_ru/_groups55_v3")
|
| 42 |
+
SR = 24000
|
| 43 |
+
TARGET = int(54.5 * SR)
|
| 44 |
+
MIN_GROUP = int(35.0 * SR)
|
| 45 |
+
MIN_UNIQUE = int(15.0 * SR)
|
| 46 |
+
FADE_MIN = int(0.002 * SR)
|
| 47 |
+
FADE_MAX = int(0.040 * SR)
|
| 48 |
+
TONE_WIN = int(0.060 * SR)
|
| 49 |
+
LATIN = re.compile(r"[a-zA-Z]")
|
| 50 |
+
|
| 51 |
+
SHORT_MIN, SHORT_MAX = int(5 * SR), int(35 * SR)
|
| 52 |
+
Q_EXTRA_PASSES = 2
|
| 53 |
+
SHORT_FRAC = 0.10
|
| 54 |
+
Q_MIN_GROUP = int(10 * SR)
|
| 55 |
+
|
| 56 |
+
# клампы семплированных пауз, сек (хвосты эмпирики не должны съедать бюджет группы)
|
| 57 |
+
# v4: клампы = p5..p95 эмпирики (v3 давал верх 1.2с — модель училась редким
|
| 58 |
+
# сверхдлинным паузам и на инференсе выдавала медиану 0.85с при норме 0.39с)
|
| 59 |
+
GAP_CLAMP = {".": (0.10, 0.90), "!": (0.10, 1.05), "?": (0.10, 1.00),
|
| 60 |
+
",": (0.06, 0.70), "none": (0.05, 0.60)}
|
| 61 |
+
|
| 62 |
+
OVERWRITE = False # --overwrite: перерендер существующих wav (фикс тона без пересборки плана)
|
| 63 |
+
|
| 64 |
+
# v10: приведение каждого клипа к -23 LUFS при рендере (--lufs). Тихая бытовая
|
| 65 |
+
# запись промпта = OOD; корпус и промпт демки нормализуются к одному уровню
|
| 66 |
+
# (стандарт NeMo/NVIDIA voice cloning). Метр создаётся лениво в каждом воркере.
|
| 67 |
+
LUFS_ON = False
|
| 68 |
+
TARGET_LUFS = -23.0
|
| 69 |
+
_meter = None
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
def lufs_gain(wav: np.ndarray) -> float:
|
| 73 |
+
global _meter
|
| 74 |
+
if len(wav) < SR // 2: # метру BS.1770 нужно >= 0.4 c
|
| 75 |
+
return 1.0
|
| 76 |
+
if _meter is None:
|
| 77 |
+
import pyloudnorm
|
| 78 |
+
_meter = pyloudnorm.Meter(SR)
|
| 79 |
+
try:
|
| 80 |
+
loud = _meter.integrated_loudness(wav)
|
| 81 |
+
except Exception: # noqa: BLE001
|
| 82 |
+
return 1.0
|
| 83 |
+
if not np.isfinite(loud) or loud < -70:
|
| 84 |
+
return 1.0
|
| 85 |
+
g = float(10 ** ((TARGET_LUFS - loud) / 20))
|
| 86 |
+
peak = float(np.abs(wav).max()) * g
|
| 87 |
+
if peak > 0.99: # true-peak защита
|
| 88 |
+
g *= 0.99 / peak
|
| 89 |
+
return min(max(g, 0.05), 20.0)
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
_hist = None # {cls: np.array длительностей}
|
| 93 |
+
|
| 94 |
+
|
| 95 |
+
def load_hist():
|
| 96 |
+
global _hist
|
| 97 |
+
h = json.load(open(ROOT / "v3" / "pause_hist.json"))
|
| 98 |
+
_hist = {k: np.asarray(v["sample"], dtype=np.float32) for k, v in h.items()}
|
| 99 |
+
return _hist
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
# медианы пауз по классу знака (v3/pause_hist.json, 4.5M пауз корпуса)
|
| 103 |
+
GAP_MED = {".": 0.39, "!": 0.41, "?": 0.42, ",": 0.27, "none": 0.11}
|
| 104 |
+
# внутригрупповой разброс: CV 0.54 — ИЗМЕРЕННАЯ вариативность пауз внутри одной
|
| 105 |
+
# записи одного диктора (lognormal sigma = sqrt(ln(1+CV^2)))
|
| 106 |
+
SIGMA_WITHIN = 0.50
|
| 107 |
+
# межгрупповой: медиана паузы '.' по спикерам гуляет 0.07..0.57с (p10..p90)
|
| 108 |
+
SIGMA_BETWEEN = 0.35
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
def gap_of(text: str, rng: random.Random, tempo: float = 1.0) -> int:
|
| 112 |
+
"""Пауза после клипа = медиана класса * ТЕМП ГРУППЫ * внутренний шум.
|
| 113 |
+
|
| 114 |
+
v3 брал семпл из ОБЩЕКОРПУСНОЙ эмпирики (CV 0.70) на каждый стык
|
| 115 |
+
независимо — внутри одной группы возникал разброс, который в реальности
|
| 116 |
+
бывает только МЕЖДУ дикторами; модель выучила «после точки может быть что
|
| 117 |
+
угодно» и на инференсе давала «то слишком коротко, то слишком долго»
|
| 118 |
+
(жалоба пользователя; замер синтеза: медиана паузы 0.85с при норме 0.39с).
|
| 119 |
+
v4 расщепляет дисперсию: tempo — один множитель на группу (междикторская
|
| 120 |
+
компонента), шум CV 0.54 — внутридикторская. Суммарно даёт корпусную
|
| 121 |
+
вариативность, но СТРУКТУРИРОВАННУЮ.
|
| 122 |
+
"""
|
| 123 |
+
tail = str(text).rstrip()[-1:]
|
| 124 |
+
cls = tail if tail in ".!?," else "none"
|
| 125 |
+
lo, hi = GAP_CLAMP[cls]
|
| 126 |
+
g = GAP_MED[cls] * tempo * rng.lognormvariate(0.0, SIGMA_WITHIN)
|
| 127 |
+
return int(min(max(g, lo), hi) * SR)
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
def group_tempo(rng: random.Random) -> float:
|
| 131 |
+
"""Логнормальный множитель темпа пауз группы (медиана 1.0, σ=0.35):
|
| 132 |
+
~80% групп в диапазоне 0.64-1.57x — межспикерная компонента."""
|
| 133 |
+
return float(min(max(rng.lognormvariate(0.0, SIGMA_BETWEEN), 0.5), 2.0))
|
| 134 |
+
|
| 135 |
+
|
| 136 |
+
def aligned_indices() -> set:
|
| 137 |
+
"""Индексы манифеста, у которых есть TextGrid (скан mfa_out, ~1 мин)."""
|
| 138 |
+
idx = set()
|
| 139 |
+
for d in MFA_OUT.iterdir():
|
| 140 |
+
if not d.is_dir():
|
| 141 |
+
continue
|
| 142 |
+
for f in d.iterdir():
|
| 143 |
+
n = f.name
|
| 144 |
+
if n.endswith(".TextGrid"):
|
| 145 |
+
try:
|
| 146 |
+
idx.add(int(n[:-9]))
|
| 147 |
+
except ValueError:
|
| 148 |
+
pass
|
| 149 |
+
return idx
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
# ---------------------------------------------------------------- план (pack)
|
| 153 |
+
def _rescue(short, pool, rng):
|
| 154 |
+
rescued, dropped = [], 0
|
| 155 |
+
uniq_total = sum(p[3] for p in pool)
|
| 156 |
+
for g in short:
|
| 157 |
+
if uniq_total < MIN_UNIQUE:
|
| 158 |
+
dropped += 1
|
| 159 |
+
continue
|
| 160 |
+
cand = [p for p in pool if p[0] not in set(g["idx"])] or list(pool)
|
| 161 |
+
rng.shuffle(cand)
|
| 162 |
+
ci = 0
|
| 163 |
+
while g["total"] < MIN_GROUP and ci < 4 * len(cand):
|
| 164 |
+
idx, path, samples, slot = cand[ci % len(cand)]
|
| 165 |
+
ci += 1
|
| 166 |
+
if g["total"] + slot > TARGET:
|
| 167 |
+
continue
|
| 168 |
+
for k, v in (("idx", idx), ("paths", path), ("samples", samples), ("slots", slot)):
|
| 169 |
+
g[k].append(v)
|
| 170 |
+
g["total"] += slot
|
| 171 |
+
g["reused"] = g.get("reused", 0) + 1
|
| 172 |
+
if g["total"] >= MIN_GROUP:
|
| 173 |
+
rescued.append(g)
|
| 174 |
+
else:
|
| 175 |
+
dropped += 1
|
| 176 |
+
return rescued, dropped
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
def _new(spk, kind="base"):
|
| 180 |
+
return {"speaker": spk, "idx": [], "paths": [], "slots": [], "samples": [],
|
| 181 |
+
"total": 0, "kind": kind}
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
def _push(g, idx, path, samples, slot):
|
| 185 |
+
g["idx"].append(idx)
|
| 186 |
+
g["paths"].append(path)
|
| 187 |
+
g["samples"].append(samples)
|
| 188 |
+
g["slots"].append(slot)
|
| 189 |
+
g["total"] += slot
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
def retempo(groups, texts, seed: int = 4242):
|
| 193 |
+
"""v4: назначить каждой группе общий множитель темпа пауз и пересчитать слоты.
|
| 194 |
+
|
| 195 |
+
Паузы внутри группы становятся согласованными (как у одного диктора в одной
|
| 196 |
+
сессии), а не независимыми выбросами из широкого распределения. Если после
|
| 197 |
+
пересчёта группа не влезает в TARGET — темп ужимается, в крайнем случае
|
| 198 |
+
дропается хвостовой клип.
|
| 199 |
+
"""
|
| 200 |
+
rng = random.Random(seed)
|
| 201 |
+
n_trim = 0
|
| 202 |
+
for g in groups:
|
| 203 |
+
tempo = group_tempo(rng)
|
| 204 |
+
for _ in range(6):
|
| 205 |
+
slots = [smp + gap_of(texts[i], rng, tempo) if i >= 0 else slot
|
| 206 |
+
for i, smp, slot in zip(g["idx"], g["samples"], g["slots"])]
|
| 207 |
+
if sum(slots) <= TARGET:
|
| 208 |
+
break
|
| 209 |
+
tempo *= 0.85
|
| 210 |
+
while sum(slots) > TARGET and len(slots) > 1:
|
| 211 |
+
for k in ("idx", "paths", "samples"):
|
| 212 |
+
g[k].pop()
|
| 213 |
+
slots.pop()
|
| 214 |
+
n_trim += 1
|
| 215 |
+
g["slots"] = slots
|
| 216 |
+
g["total"] = sum(slots)
|
| 217 |
+
g["tempo"] = round(tempo, 3)
|
| 218 |
+
print(f"retempo: групп {len(groups)}, обрезано хвостовых клипов {n_trim}")
|
| 219 |
+
return [g for g in groups if g["total"] >= 5 * SR and g["idx"]]
|
| 220 |
+
|
| 221 |
+
|
| 222 |
+
def tempo_key(path: str) -> str:
|
| 223 |
+
"""v6: версия темпа клипа (orig / slow / fast).
|
| 224 |
+
|
| 225 |
+
Замер на v5: модель почти не копирует темп промпта (наклон 0.34). Одна из
|
| 226 |
+
причин — 18.8% групп СМЕШИВАЛИ оригиналы с темпо-аугментированными копиями:
|
| 227 |
+
промпт мог быть 0.75x, а продолжение 1.3x, т.е. данные прямо учили, что
|
| 228 |
+
темп промпта НЕ предсказывает темп речи. Группируем по (спикер, версия) —
|
| 229 |
+
внутри группы темп однороден, связь промпт->продолжение становится честной.
|
| 230 |
+
"""
|
| 231 |
+
if "_tempo_aug" not in path:
|
| 232 |
+
return "orig"
|
| 233 |
+
return "slow" if "_slow" in path else "fast"
|
| 234 |
+
|
| 235 |
+
|
| 236 |
+
# v7: границы темпо-корзин, слог/с (clip_sps.npy: медиана 5.22, p10 3.76, p90 6.78).
|
| 237 |
+
# Замер на v6: 60% групп имели разброс темпа >=2 слог/с ВНУТРИ себя — промпт (начало
|
| 238 |
+
# группы) и продолжение записаны с разной скоростью, поэтому связь «темп промпта ->
|
| 239 |
+
# темп речи» в данных отсутствовала (наклон копирования 0.15-0.34 во всех версиях).
|
| 240 |
+
TEMPO_EDGES = (4.3, 5.2, 6.1)
|
| 241 |
+
|
| 242 |
+
|
| 243 |
+
def tempo_bucket(sps: float) -> str:
|
| 244 |
+
if not np.isfinite(sps):
|
| 245 |
+
return "na"
|
| 246 |
+
return str(int(np.searchsorted(TEMPO_EDGES, sps)))
|
| 247 |
+
|
| 248 |
+
|
| 249 |
+
def pack_groups(df, aligned: set, seed: int = 42, clip_sps=None):
|
| 250 |
+
rng = random.Random(seed)
|
| 251 |
+
groups, n_lat, n_noal, n_rescued, n_dropped, n_reused = [], 0, 0, 0, 0, 0
|
| 252 |
+
texts = df.text.astype(str)
|
| 253 |
+
by_spk_pool = {}
|
| 254 |
+
|
| 255 |
+
key = df.speaker.astype(str) + "|" + df.audio_path.map(tempo_key)
|
| 256 |
+
if clip_sps is not None:
|
| 257 |
+
key = key + "|" + pd.Series(
|
| 258 |
+
[tempo_bucket(s) for s in clip_sps[: len(df)]], index=df.index
|
| 259 |
+
)
|
| 260 |
+
df = df.assign(_spk_tempo=key)
|
| 261 |
+
for spk_t, sub in df.groupby("_spk_tempo", sort=False):
|
| 262 |
+
spk = spk_t.split("|")[0]
|
| 263 |
+
cur, short, pool = None, [], []
|
| 264 |
+
for row in sub.itertuples():
|
| 265 |
+
if int(row.Index) not in aligned:
|
| 266 |
+
n_noal += 1
|
| 267 |
+
continue
|
| 268 |
+
if LATIN.search(row.text):
|
| 269 |
+
n_lat += 1
|
| 270 |
+
continue
|
| 271 |
+
samples = int(round(row.duration * SR))
|
| 272 |
+
slot = samples + gap_of(row.text, rng)
|
| 273 |
+
if samples <= 0 or slot > TARGET:
|
| 274 |
+
continue
|
| 275 |
+
pool.append((int(row.Index), row.audio_path, samples, slot))
|
| 276 |
+
if cur is None or cur["total"] + slot > TARGET:
|
| 277 |
+
if cur:
|
| 278 |
+
(groups if cur["total"] >= MIN_GROUP else short).append(cur)
|
| 279 |
+
cur = _new(spk)
|
| 280 |
+
_push(cur, int(row.Index), row.audio_path, samples, slot)
|
| 281 |
+
if cur:
|
| 282 |
+
(groups if cur["total"] >= MIN_GROUP else short).append(cur)
|
| 283 |
+
if short:
|
| 284 |
+
rescued, dropped = _rescue(short, pool, rng)
|
| 285 |
+
n_rescued += len(rescued)
|
| 286 |
+
n_dropped += dropped
|
| 287 |
+
n_reused += sum(g.get("reused", 0) for g in rescued)
|
| 288 |
+
groups.extend(rescued)
|
| 289 |
+
by_spk_pool[spk_t] = pool
|
| 290 |
+
n_base = len(groups)
|
| 291 |
+
print(f"базовых групп: {n_base}; латиница={n_lat}, без выравнивания={n_noal}, "
|
| 292 |
+
f"спасено={n_rescued} (+{n_reused} переисп.), дропнуто={n_dropped}")
|
| 293 |
+
|
| 294 |
+
# --- оверсемпл вопросов ---
|
| 295 |
+
n_q = 0
|
| 296 |
+
for spk_t, pool in by_spk_pool.items():
|
| 297 |
+
spk = spk_t.rsplit("|", 1)[0]
|
| 298 |
+
qs = [p for p in pool if "?" in texts[p[0]]]
|
| 299 |
+
if not qs:
|
| 300 |
+
continue
|
| 301 |
+
for _ in range(Q_EXTRA_PASSES):
|
| 302 |
+
order = qs[:]
|
| 303 |
+
rng.shuffle(order)
|
| 304 |
+
cur = None
|
| 305 |
+
for idx, path, samples, slot in order:
|
| 306 |
+
if cur is None or cur["total"] + slot > TARGET:
|
| 307 |
+
if cur and cur["total"] >= Q_MIN_GROUP:
|
| 308 |
+
groups.append(cur)
|
| 309 |
+
cur = _new(spk, "question")
|
| 310 |
+
_push(cur, idx, path, samples, slot)
|
| 311 |
+
if cur and cur["total"] >= Q_MIN_GROUP:
|
| 312 |
+
groups.append(cur)
|
| 313 |
+
n_q = len(groups) - n_base
|
| 314 |
+
|
| 315 |
+
# --- короткие группы ---
|
| 316 |
+
spk_list = [s for s, p in by_spk_pool.items() if p] # ключи "спикер|версия"
|
| 317 |
+
n_short_target = int(SHORT_FRAC * n_base)
|
| 318 |
+
for _ in range(n_short_target):
|
| 319 |
+
spk_t = rng.choice(spk_list)
|
| 320 |
+
pool = by_spk_pool[spk_t]
|
| 321 |
+
cur = _new(spk_t.rsplit("|", 1)[0], "short")
|
| 322 |
+
target_len = rng.randint(SHORT_MIN, SHORT_MAX)
|
| 323 |
+
for _try in range(6):
|
| 324 |
+
idx, path, samples, slot = pool[rng.randrange(len(pool))]
|
| 325 |
+
if cur["total"] + slot > min(target_len + 3 * SR, TARGET):
|
| 326 |
+
if cur["total"] >= SHORT_MIN:
|
| 327 |
+
break
|
| 328 |
+
continue
|
| 329 |
+
_push(cur, idx, path, samples, slot)
|
| 330 |
+
if cur["total"] >= target_len:
|
| 331 |
+
break
|
| 332 |
+
if cur["idx"]:
|
| 333 |
+
groups.append(cur)
|
| 334 |
+
print(f"вопросных групп: +{n_q}, коротких: +{len(groups) - n_base - n_q}")
|
| 335 |
+
return retempo(groups, texts)
|
| 336 |
+
|
| 337 |
+
|
| 338 |
+
# ------------------------------------------------------------- рендер (wav)
|
| 339 |
+
def _cos_ramp(n: int) -> np.ndarray:
|
| 340 |
+
return (0.5 - 0.5 * np.cos(np.linspace(0, np.pi, n, dtype=np.float32)))
|
| 341 |
+
|
| 342 |
+
|
| 343 |
+
def adaptive_fades(wav: np.ndarray) -> np.ndarray:
|
| 344 |
+
"""Фейды внутри краевой тишины клипа: до 40 мс, речь не трогаем."""
|
| 345 |
+
peak = float(np.max(np.abs(wav))) + 1e-9
|
| 346 |
+
loud = np.abs(wav) > max(0.02 * peak, 1e-4)
|
| 347 |
+
if not loud.any():
|
| 348 |
+
return wav
|
| 349 |
+
first = int(np.argmax(loud))
|
| 350 |
+
last = len(wav) - int(np.argmax(loud[::-1]))
|
| 351 |
+
fi = min(max(first, FADE_MIN), FADE_MAX, len(wav))
|
| 352 |
+
fo = min(max(len(wav) - last, FADE_MIN), FADE_MAX, len(wav))
|
| 353 |
+
wav[:fi] *= _cos_ramp(fi)
|
| 354 |
+
wav[-fo:] *= _cos_ramp(fo)[::-1]
|
| 355 |
+
return wav
|
| 356 |
+
|
| 357 |
+
|
| 358 |
+
def room_tone(wav: np.ndarray) -> np.ndarray:
|
| 359 |
+
"""Самое тихое 120-мс окно клипа (шаблон спектра); тихих нет — глушим до -46 dBFS."""
|
| 360 |
+
win = 2 * TONE_WIN
|
| 361 |
+
if len(wav) < 2 * win:
|
| 362 |
+
return np.zeros(win, dtype=np.float32)
|
| 363 |
+
hop = win // 2
|
| 364 |
+
n = (len(wav) - win) // hop
|
| 365 |
+
view = np.lib.stride_tricks.sliding_window_view(wav, win)[::hop][:n]
|
| 366 |
+
rms = np.sqrt((view ** 2).mean(axis=1))
|
| 367 |
+
k = int(np.argmin(rms))
|
| 368 |
+
tone = view[k].copy()
|
| 369 |
+
if rms[k] > 5e-3:
|
| 370 |
+
tone *= 5e-3 / rms[k]
|
| 371 |
+
return tone
|
| 372 |
+
|
| 373 |
+
|
| 374 |
+
_N_FFT = 512
|
| 375 |
+
_HOP = 256
|
| 376 |
+
_HANN = np.hanning(_N_FFT).astype(np.float32)
|
| 377 |
+
|
| 378 |
+
|
| 379 |
+
def synth_tone(template: np.ndarray, length: int, seed: int) -> np.ndarray:
|
| 380 |
+
"""Стационарный шум со спектральной огибающей шаблона — БЕЗ зацикливания.
|
| 381 |
+
|
| 382 |
+
v3 первой версии тайлил 60-мс окно зеркально: период ~8 Гц слышен как
|
| 383 |
+
«вертолёт», модель выучила текстуру (жалоба юзера). Здесь — средняя
|
| 384 |
+
STFT-магнитуда шаблона + случайные фазы на каждый кадр + overlap-add:
|
| 385 |
+
цвет комнаты сохранён, повторов нет в принципе.
|
| 386 |
+
"""
|
| 387 |
+
if length <= 0:
|
| 388 |
+
return np.zeros(0, dtype=np.float32)
|
| 389 |
+
rng = np.random.default_rng(seed)
|
| 390 |
+
if len(template) < _N_FFT or not np.any(template):
|
| 391 |
+
return np.zeros(length, dtype=np.float32)
|
| 392 |
+
n = (len(template) - _N_FFT) // _HOP + 1
|
| 393 |
+
frames = np.lib.stride_tricks.sliding_window_view(template, _N_FFT)[::_HOP][:n]
|
| 394 |
+
mag = np.abs(np.fft.rfft(frames * _HANN, axis=1)).mean(axis=0)
|
| 395 |
+
|
| 396 |
+
n_frames = length // _HOP + 3
|
| 397 |
+
phases = rng.uniform(0, 2 * np.pi, size=(n_frames, len(mag)))
|
| 398 |
+
sig_frames = np.fft.irfft(mag * np.exp(1j * phases), n=_N_FFT, axis=1).real
|
| 399 |
+
sig_frames *= _HANN
|
| 400 |
+
out = np.zeros(n_frames * _HOP + _N_FFT, dtype=np.float32)
|
| 401 |
+
for i in range(n_frames): # OLA (hann, hop=1/2 -> COLA)
|
| 402 |
+
out[i * _HOP:i * _HOP + _N_FFT] += sig_frames[i]
|
| 403 |
+
out = out[_N_FFT: _N_FFT + length]
|
| 404 |
+
t_rms = float(np.sqrt((template ** 2).mean()))
|
| 405 |
+
o_rms = float(np.sqrt((out ** 2).mean())) + 1e-12
|
| 406 |
+
return (out * (t_rms / o_rms)).astype(np.float32)
|
| 407 |
+
|
| 408 |
+
|
| 409 |
+
def fill_tone(buf: np.ndarray, start: int, end: int, tone: np.ndarray, seed: int):
|
| 410 |
+
"""Заполняет [start:end) синтезированным комнатным тоном с 5-мс рампами."""
|
| 411 |
+
if end <= start or not len(tone):
|
| 412 |
+
return
|
| 413 |
+
buf[start:end] = synth_tone(tone, end - start, seed)
|
| 414 |
+
r = min(int(0.005 * SR), (end - start) // 2)
|
| 415 |
+
if r > 0:
|
| 416 |
+
buf[start:start + r] *= _cos_ramp(r)
|
| 417 |
+
buf[end - r:end] *= _cos_ramp(r)[::-1]
|
| 418 |
+
|
| 419 |
+
|
| 420 |
+
def write_group(task):
|
| 421 |
+
gi, g = task
|
| 422 |
+
out = GROUP_DIR / f"g{gi:06d}.wav"
|
| 423 |
+
if not OVERWRITE and out.exists():
|
| 424 |
+
try:
|
| 425 |
+
if sf.info(str(out)).frames == int(55 * SR):
|
| 426 |
+
return gi, str(out), "skip"
|
| 427 |
+
except Exception:
|
| 428 |
+
out.unlink(missing_ok=True)
|
| 429 |
+
buf = np.zeros(int(55 * SR), dtype=np.float32)
|
| 430 |
+
pos = 0
|
| 431 |
+
try:
|
| 432 |
+
tone = np.zeros(2 * TONE_WIN, dtype=np.float32)
|
| 433 |
+
for j, (path, samples, slot) in enumerate(zip(g["paths"], g["samples"], g["slots"])):
|
| 434 |
+
wav, sr = sf.read(path, dtype="float32")
|
| 435 |
+
assert sr == SR, f"{path}: sr={sr}"
|
| 436 |
+
if wav.ndim > 1:
|
| 437 |
+
wav = wav.mean(axis=1)
|
| 438 |
+
wav = wav[:samples].copy()
|
| 439 |
+
if LUFS_ON:
|
| 440 |
+
wav *= lufs_gain(wav)
|
| 441 |
+
wav = adaptive_fades(wav)
|
| 442 |
+
buf[pos:pos + len(wav)] = wav
|
| 443 |
+
tone = room_tone(wav)
|
| 444 |
+
fill_tone(buf, pos + len(wav), pos + slot, tone, 100_000 + gi * 64 + j)
|
| 445 |
+
pos += slot
|
| 446 |
+
fill_tone(buf, pos, len(buf), tone, 100_000 + gi * 64 + 63) # хвост группы
|
| 447 |
+
sf.write(out, buf, SR, subtype="PCM_16")
|
| 448 |
+
return gi, str(out), "ok"
|
| 449 |
+
except Exception as e: # noqa: BLE001
|
| 450 |
+
out.unlink(missing_ok=True)
|
| 451 |
+
return gi, "", f"fail:{e}"
|
| 452 |
+
|
| 453 |
+
|
| 454 |
+
def main():
|
| 455 |
+
ap = argparse.ArgumentParser()
|
| 456 |
+
ap.add_argument("--plan-only", action="store_true")
|
| 457 |
+
ap.add_argument("--workers", type=int, default=24)
|
| 458 |
+
ap.add_argument("--chunk-size", type=int, default=70_000)
|
| 459 |
+
ap.add_argument("--overwrite", action="store_true",
|
| 460 |
+
help="перерендерить wav даже если файл на месте (план не меняется)")
|
| 461 |
+
global OVERWRITE, MFA_OUT, GROUP_DIR
|
| 462 |
+
ap.add_argument("--mfa-out", default="mfa_out", help="каталог TextGrid (v4: mfa_out_v4)")
|
| 463 |
+
ap.add_argument("--out-groups", default="groups_v3.json")
|
| 464 |
+
ap.add_argument("--group-dir", default="/mnt/data/audio_data/voxtream_ru/_groups55_v3")
|
| 465 |
+
ap.add_argument("--parquet-prefix", default="grouped_v3_chunk")
|
| 466 |
+
ap.add_argument("--tempo-buckets", action="store_true",
|
| 467 |
+
help="v7: группы однородны по темпу (clip_sps.npy)")
|
| 468 |
+
ap.add_argument("--clip-sps", default="clip_sps.npy",
|
| 469 |
+
help="файл SPS по клипам для темпо-корзин")
|
| 470 |
+
ap.add_argument("--lufs", action="store_true",
|
| 471 |
+
help="v10: каждый клип -> -23 LUFS при рендере")
|
| 472 |
+
args = ap.parse_args()
|
| 473 |
+
global LUFS_ON
|
| 474 |
+
OVERWRITE = args.overwrite
|
| 475 |
+
LUFS_ON = args.lufs
|
| 476 |
+
MFA_OUT = ROOT / args.mfa_out
|
| 477 |
+
GROUP_DIR = Path(args.group_dir)
|
| 478 |
+
if LUFS_ON:
|
| 479 |
+
print(f"LUFS-нормализация ВКЛ: клипы -> {TARGET_LUFS} LUFS")
|
| 480 |
+
|
| 481 |
+
load_hist()
|
| 482 |
+
print("скан выравниваний…", flush=True)
|
| 483 |
+
aligned = aligned_indices()
|
| 484 |
+
print(f"выровненных клипов: {len(aligned)}", flush=True)
|
| 485 |
+
|
| 486 |
+
df = pd.read_csv(ROOT / "manifest.csv", sep="|", low_memory=False)
|
| 487 |
+
clip_sps = None
|
| 488 |
+
if args.tempo_buckets:
|
| 489 |
+
clip_sps = np.load(ROOT / args.clip_sps)
|
| 490 |
+
print(f"темпо-корзины ВКЛ: границы {TEMPO_EDGES} слог/с")
|
| 491 |
+
groups = pack_groups(df, aligned, clip_sps=clip_sps)
|
| 492 |
+
total_h = sum(g["total"] for g in groups) / SR / 3600
|
| 493 |
+
used = len({i for g in groups for i in g["idx"]})
|
| 494 |
+
print(f"групп: {len(groups)}, уникальных клипов: {used}/{len(df)}, ~{total_h:.1f} ч слотов")
|
| 495 |
+
json.dump(groups, open(ROOT / args.out_groups, "w"), ensure_ascii=False)
|
| 496 |
+
if args.plan_only:
|
| 497 |
+
return
|
| 498 |
+
|
| 499 |
+
GROUP_DIR.mkdir(parents=True, exist_ok=True)
|
| 500 |
+
fails = 0
|
| 501 |
+
with mp.Pool(args.workers) as pool:
|
| 502 |
+
for gi, path, status in pool.imap(write_group, enumerate(groups), chunksize=16):
|
| 503 |
+
groups[gi]["group_wav"] = path
|
| 504 |
+
fails += status.startswith("fail")
|
| 505 |
+
if (gi + 1) % 10_000 == 0:
|
| 506 |
+
print(f" {gi + 1}/{len(groups)} (fail={fails})", flush=True)
|
| 507 |
+
groups = [g for g in groups if g.get("group_wav")]
|
| 508 |
+
json.dump(groups, open(ROOT / args.out_groups, "w"), ensure_ascii=False)
|
| 509 |
+
print(f"записано групп: {len(groups)} (fail={fails})")
|
| 510 |
+
|
| 511 |
+
paths = [g["group_wav"] for g in groups]
|
| 512 |
+
for ci in range(0, len(paths), args.chunk_size):
|
| 513 |
+
p = ROOT / f"{args.parquet_prefix}{ci // args.chunk_size}.parquet"
|
| 514 |
+
pd.DataFrame({"paths": paths[ci:ci + args.chunk_size]}).to_parquet(p, index=False)
|
| 515 |
+
print(f"{p.name}: {min(args.chunk_size, len(paths) - ci)} путей")
|
| 516 |
+
|
| 517 |
+
|
| 518 |
+
if __name__ == "__main__":
|
| 519 |
+
main()
|
demo/configs/app.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"min_chunk_sec": 0.01,
|
| 3 |
+
"fade_out_sec": 0.1,
|
| 4 |
+
"plot_window_sec": 10.0,
|
| 5 |
+
"visual_update_sec": 0.25,
|
| 6 |
+
"future_phone_limit": 25,
|
| 7 |
+
"plot_width": 1000,
|
| 8 |
+
"plot_height": 224,
|
| 9 |
+
"plot_left": 74,
|
| 10 |
+
"plot_right": 22,
|
| 11 |
+
"plot_top": 50,
|
| 12 |
+
"plot_bottom": 42,
|
| 13 |
+
"plot_y_max": 7,
|
| 14 |
+
"plot_y_tick": 1,
|
| 15 |
+
"plot_x_tick_sec": 1,
|
| 16 |
+
"audio_stream_start_delay_sec": 0.12,
|
| 17 |
+
"audio_stream_sample_rate": 24000,
|
| 18 |
+
"speaking_rate_min": 1.0,
|
| 19 |
+
"speaking_rate_max": 7.0,
|
| 20 |
+
"speaking_rate_step": 0.1,
|
| 21 |
+
"speaking_rate_default": 4.0,
|
| 22 |
+
"min_streaming_rtf": 0.95
|
| 23 |
+
}
|
demo/configs/generator.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"sil_token": 120,
|
| 3 |
+
"bos_token": 123,
|
| 4 |
+
"eos_token": 124,
|
| 5 |
+
"unk_token": 122,
|
| 6 |
+
"eop_token": 122,
|
| 7 |
+
"num_codebooks": 16,
|
| 8 |
+
"num_phones_per_frame": 2,
|
| 9 |
+
"audio_delay_frames": 1,
|
| 10 |
+
"temperature": 0.9,
|
| 11 |
+
"topk": 5,
|
| 12 |
+
"top_p": 0.9,
|
| 13 |
+
"max_audio_length_ms": 60000,
|
| 14 |
+
"model_repo": "herimor/voxtream2",
|
| 15 |
+
"model_name": "model.safetensors",
|
| 16 |
+
"model_config_name": "config.json",
|
| 17 |
+
"mimi_sr": 24000,
|
| 18 |
+
"mimi_vocab_size": 2048,
|
| 19 |
+
"mimi_frame_ms": 80,
|
| 20 |
+
"mimi_repo": "kyutai/moshiko-pytorch-bf16",
|
| 21 |
+
"mimi_name": "tokenizer-e351c8d8-checkpoint125.safetensors",
|
| 22 |
+
"spk_enc_sr": 16000,
|
| 23 |
+
"spk_enc_repo": "IDRnD/ReDimNet",
|
| 24 |
+
"spk_enc_model": "ReDimNet",
|
| 25 |
+
"spk_enc_model_name": "M",
|
| 26 |
+
"spk_enc_train_type": "ft_mix",
|
| 27 |
+
"spk_enc_dataset": "vb2+vox2+cnc",
|
| 28 |
+
"phoneme_dict_name": "phoneme_to_token.json",
|
| 29 |
+
"max_prompt_sec": 20,
|
| 30 |
+
"min_prompt_sec": 1,
|
| 31 |
+
"max_phone_tokens": 2000,
|
| 32 |
+
"cache_prompt": false,
|
| 33 |
+
"cfg_gamma": 1.5,
|
| 34 |
+
"cfg_ac_gamma": 3.0,
|
| 35 |
+
"text_context": " context",
|
| 36 |
+
"text_context_length": 18,
|
| 37 |
+
"spk_proj_weight": 1.5,
|
| 38 |
+
"audio_pad_token": 2049,
|
| 39 |
+
"enhance_prompt": false,
|
| 40 |
+
"sidon_se_reload_model": false,
|
| 41 |
+
"reset_streaming_state": false,
|
| 42 |
+
"hf_token": null,
|
| 43 |
+
"apply_vad": false,
|
| 44 |
+
"min_speech_seg_sec": 0.3,
|
| 45 |
+
"min_look_ahead_phones": 3,
|
| 46 |
+
"phonemizer": "espeak",
|
| 47 |
+
"spk_rate_window_sec": 3.0,
|
| 48 |
+
"frame_repeat_counter": 25,
|
| 49 |
+
"punct_map": {
|
| 50 |
+
".": 117,
|
| 51 |
+
",": 118,
|
| 52 |
+
"?": 119,
|
| 53 |
+
"!": 121
|
| 54 |
+
},
|
| 55 |
+
"phoneme_index_map":{
|
| 56 |
+
"0": [0, 1], "1": [0, 2], "2": [1, 1], "3": [1, 2], "4": [2, 1], "5": [2, 2]
|
| 57 |
+
}
|
| 58 |
+
}
|
demo/configs/speaking_rate.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"1": {
|
| 3 |
+
"duration_state": [
|
| 4 |
+
52, 15, 1, 2, 1, 3
|
| 5 |
+
],
|
| 6 |
+
"weight": 3.0,
|
| 7 |
+
"cfg_gamma": 1.25
|
| 8 |
+
},
|
| 9 |
+
"2": {
|
| 10 |
+
"duration_state": [
|
| 11 |
+
44, 24, 2, 2, 1, 1
|
| 12 |
+
],
|
| 13 |
+
"weight": 3.0,
|
| 14 |
+
"cfg_gamma": 1.25
|
| 15 |
+
},
|
| 16 |
+
"3": {
|
| 17 |
+
"duration_state": [
|
| 18 |
+
24, 23, 2, 3, 1, 1
|
| 19 |
+
],
|
| 20 |
+
"weight": 5.0,
|
| 21 |
+
"cfg_gamma": 1.5
|
| 22 |
+
},
|
| 23 |
+
"4": {
|
| 24 |
+
"duration_state": [
|
| 25 |
+
40, 59, 5, 12, 1, 2
|
| 26 |
+
],
|
| 27 |
+
"weight": 5.0,
|
| 28 |
+
"cfg_gamma": 1.5
|
| 29 |
+
},
|
| 30 |
+
"5": {
|
| 31 |
+
"duration_state": [
|
| 32 |
+
21, 43, 4, 13, 1, 2
|
| 33 |
+
],
|
| 34 |
+
"weight": 7.0,
|
| 35 |
+
"cfg_gamma": 2.0
|
| 36 |
+
},
|
| 37 |
+
"6": {
|
| 38 |
+
"duration_state": [
|
| 39 |
+
27, 69, 6, 30, 2, 7
|
| 40 |
+
],
|
| 41 |
+
"weight": 7.0,
|
| 42 |
+
"cfg_gamma": 2.0
|
| 43 |
+
},
|
| 44 |
+
"7": {
|
| 45 |
+
"duration_state": [
|
| 46 |
+
6, 18, 2, 9, 1, 3
|
| 47 |
+
],
|
| 48 |
+
"weight": 10.0,
|
| 49 |
+
"cfg_gamma": 2.5
|
| 50 |
+
}
|
| 51 |
+
}
|
demo/configs/speaking_rate_ru.json
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"1": {
|
| 3 |
+
"duration_state": [
|
| 4 |
+
73,
|
| 5 |
+
23,
|
| 6 |
+
2,
|
| 7 |
+
3,
|
| 8 |
+
1,
|
| 9 |
+
1
|
| 10 |
+
],
|
| 11 |
+
"weight": 8.0,
|
| 12 |
+
"cfg_gamma": 1.25
|
| 13 |
+
},
|
| 14 |
+
"2": {
|
| 15 |
+
"duration_state": [
|
| 16 |
+
60,
|
| 17 |
+
33,
|
| 18 |
+
2,
|
| 19 |
+
5,
|
| 20 |
+
1,
|
| 21 |
+
1
|
| 22 |
+
],
|
| 23 |
+
"weight": 8.0,
|
| 24 |
+
"cfg_gamma": 1.25
|
| 25 |
+
},
|
| 26 |
+
"3": {
|
| 27 |
+
"duration_state": [
|
| 28 |
+
43,
|
| 29 |
+
47,
|
| 30 |
+
3,
|
| 31 |
+
7,
|
| 32 |
+
1,
|
| 33 |
+
1
|
| 34 |
+
],
|
| 35 |
+
"weight": 8.0,
|
| 36 |
+
"cfg_gamma": 1.5
|
| 37 |
+
},
|
| 38 |
+
"4": {
|
| 39 |
+
"duration_state": [
|
| 40 |
+
29,
|
| 41 |
+
56,
|
| 42 |
+
3,
|
| 43 |
+
11,
|
| 44 |
+
1,
|
| 45 |
+
1
|
| 46 |
+
],
|
| 47 |
+
"weight": 5.0,
|
| 48 |
+
"cfg_gamma": 1.5
|
| 49 |
+
},
|
| 50 |
+
"5": {
|
| 51 |
+
"duration_state": [
|
| 52 |
+
21,
|
| 53 |
+
58,
|
| 54 |
+
4,
|
| 55 |
+
17,
|
| 56 |
+
1,
|
| 57 |
+
1
|
| 58 |
+
],
|
| 59 |
+
"weight": 7.0,
|
| 60 |
+
"cfg_gamma": 2.0
|
| 61 |
+
},
|
| 62 |
+
"6": {
|
| 63 |
+
"duration_state": [
|
| 64 |
+
14,
|
| 65 |
+
57,
|
| 66 |
+
4,
|
| 67 |
+
24,
|
| 68 |
+
1,
|
| 69 |
+
1
|
| 70 |
+
],
|
| 71 |
+
"weight": 8.0,
|
| 72 |
+
"cfg_gamma": 2.0
|
| 73 |
+
},
|
| 74 |
+
"7": {
|
| 75 |
+
"duration_state": [
|
| 76 |
+
9,
|
| 77 |
+
51,
|
| 78 |
+
3,
|
| 79 |
+
34,
|
| 80 |
+
1,
|
| 81 |
+
3
|
| 82 |
+
],
|
| 83 |
+
"weight": 8.0,
|
| 84 |
+
"cfg_gamma": 2.5
|
| 85 |
+
}
|
| 86 |
+
}
|
demo/examples_ru.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"examples": []}
|
demo/packages.txt
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
espeak-ng
|
demo/requirements.txt
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Пакет voxtream вендорен в спейс (модифицированная версия под RU-модель:
|
| 2 |
+
# классовые лимиты удержания, SRC-режим, VOXTREAM_LA) — с GitHub НЕ ставить.
|
| 3 |
+
torch>=2.4,<2.9
|
| 4 |
+
torchaudio>=2.4,<2.9
|
| 5 |
+
torchtune==0.4.0
|
| 6 |
+
torchao==0.9.0
|
| 7 |
+
moshi>=0.2.13
|
| 8 |
+
transformers==4.50.0
|
| 9 |
+
huggingface_hub==0.28.1
|
| 10 |
+
tokenizers<0.22
|
| 11 |
+
g2p-en==2.1.0
|
| 12 |
+
librosa==0.11.0
|
| 13 |
+
soundfile==0.13.1
|
| 14 |
+
inflect==7.5.0
|
| 15 |
+
nltk==3.9.1
|
| 16 |
+
gradio==4.44.1
|
| 17 |
+
gradio_client==1.3.0
|
| 18 |
+
starlette==0.52.1
|
| 19 |
+
pydantic==2.10.6
|
| 20 |
+
silero-vad==6.2.0
|
| 21 |
+
ruaccent
|
| 22 |
+
runorm==1.1
|
| 23 |
+
pyloudnorm
|
| 24 |
+
numpy
|
| 25 |
+
pandas
|
demo/runorm_cache/.locks/models--RUNorm--RUNorm-normalizer-medium/196088ae0adbafc572d99e62d0090a4a25a0bb305493ff454a7f9fb2171c1566.lock
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/added_tokens.json
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/chat_template.jinja
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/model.safetensors
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/model.safetensors.index.json
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/.no_exist/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer.model
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/1e39d930e09fd435aa8be1182640092b0581e3ca
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": "<s>",
|
| 3 |
+
"eos_token": "</s>",
|
| 4 |
+
"pad_token": "<pad>"
|
| 5 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/23440d6773792a7758b5aab66878ba5f11f56f81235394a3856718c82e2a8f1e
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:23440d6773792a7758b5aab66878ba5f11f56f81235394a3856718c82e2a8f1e
|
| 3 |
+
size 16811983
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/49f05ecdcda8ba78b292f78ba2f83016d348cf62
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"decoder_start_token_id": 0,
|
| 4 |
+
"eos_token_id": 1,
|
| 5 |
+
"pad_token_id": 0,
|
| 6 |
+
"transformers_version": "4.28.1"
|
| 7 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/88ae2e52858aba88b4e935b2e5689ab4dc49bae6
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"clean_up_tokenization_spaces": true,
|
| 3 |
+
"extra_ids": 0,
|
| 4 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 5 |
+
"tokenizer_class": "PreTrainedTokenizerFast"
|
| 6 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/a5a347155014013f45c8adb3efaebf40d68142ed
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "maximxls/text-normalization-ru-terrible",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"T5ForConditionalGeneration"
|
| 5 |
+
],
|
| 6 |
+
"d_ff": 1024,
|
| 7 |
+
"d_kv": 64,
|
| 8 |
+
"d_model": 256,
|
| 9 |
+
"decoder_start_token_id": 0,
|
| 10 |
+
"dense_act_fn": "gelu_new",
|
| 11 |
+
"dropout_rate": 0.0,
|
| 12 |
+
"eos_token_id": 1,
|
| 13 |
+
"feed_forward_proj": "gated-gelu",
|
| 14 |
+
"initializer_factor": 1.0,
|
| 15 |
+
"is_encoder_decoder": true,
|
| 16 |
+
"is_gated_act": true,
|
| 17 |
+
"layer_norm_epsilon": 1e-06,
|
| 18 |
+
"model_type": "t5",
|
| 19 |
+
"num_decoder_layers": 3,
|
| 20 |
+
"num_heads": 4,
|
| 21 |
+
"num_layers": 3,
|
| 22 |
+
"pad_token_id": 0,
|
| 23 |
+
"relative_attention_max_distance": 128,
|
| 24 |
+
"relative_attention_num_buckets": 32,
|
| 25 |
+
"torch_dtype": "bfloat16",
|
| 26 |
+
"transformers_version": "4.28.1",
|
| 27 |
+
"use_cache": true,
|
| 28 |
+
"vocab_size": 5120
|
| 29 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/c96c58140281ac3b8e2993004ebc727ff5b22171
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/blobs/e33efa79e656bdd0972beb15a07b11b74df0df1368308bd12ec6b02337887040
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e33efa79e656bdd0972beb15a07b11b74df0df1368308bd12ec6b02337887040
|
| 3 |
+
size 16795208
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/refs/main
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
b130ae67db4b209babec461767bcd2ace74fe88a
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/refs/refs/pr/1
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
c282bffa4b5a328a29883541fc4f8784f1ffe093
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/config.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "maximxls/text-normalization-ru-terrible",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"T5ForConditionalGeneration"
|
| 5 |
+
],
|
| 6 |
+
"d_ff": 1024,
|
| 7 |
+
"d_kv": 64,
|
| 8 |
+
"d_model": 256,
|
| 9 |
+
"decoder_start_token_id": 0,
|
| 10 |
+
"dense_act_fn": "gelu_new",
|
| 11 |
+
"dropout_rate": 0.0,
|
| 12 |
+
"eos_token_id": 1,
|
| 13 |
+
"feed_forward_proj": "gated-gelu",
|
| 14 |
+
"initializer_factor": 1.0,
|
| 15 |
+
"is_encoder_decoder": true,
|
| 16 |
+
"is_gated_act": true,
|
| 17 |
+
"layer_norm_epsilon": 1e-06,
|
| 18 |
+
"model_type": "t5",
|
| 19 |
+
"num_decoder_layers": 3,
|
| 20 |
+
"num_heads": 4,
|
| 21 |
+
"num_layers": 3,
|
| 22 |
+
"pad_token_id": 0,
|
| 23 |
+
"relative_attention_max_distance": 128,
|
| 24 |
+
"relative_attention_num_buckets": 32,
|
| 25 |
+
"torch_dtype": "bfloat16",
|
| 26 |
+
"transformers_version": "4.28.1",
|
| 27 |
+
"use_cache": true,
|
| 28 |
+
"vocab_size": 5120
|
| 29 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/generation_config.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"decoder_start_token_id": 0,
|
| 4 |
+
"eos_token_id": 1,
|
| 5 |
+
"pad_token_id": 0,
|
| 6 |
+
"transformers_version": "4.28.1"
|
| 7 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/pytorch_model.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:23440d6773792a7758b5aab66878ba5f11f56f81235394a3856718c82e2a8f1e
|
| 3 |
+
size 16811983
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/special_tokens_map.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": "<s>",
|
| 3 |
+
"eos_token": "</s>",
|
| 4 |
+
"pad_token": "<pad>"
|
| 5 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/b130ae67db4b209babec461767bcd2ace74fe88a/tokenizer_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"clean_up_tokenization_spaces": true,
|
| 3 |
+
"extra_ids": 0,
|
| 4 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 5 |
+
"tokenizer_class": "PreTrainedTokenizerFast"
|
| 6 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-kirillizator/snapshots/c282bffa4b5a328a29883541fc4f8784f1ffe093/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e33efa79e656bdd0972beb15a07b11b74df0df1368308bd12ec6b02337887040
|
| 3 |
+
size 16795208
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/chat_template.jinja
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/model.safetensors
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/.no_exist/3fdb93344da77fbd75821e2fdd6307df3a0a1c96/model.safetensors.index.json
ADDED
|
File without changes
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/010ca0a172e0d662f661ff9b6f9b3323c1272ec4
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"decoder_start_token_id": 0,
|
| 4 |
+
"eos_token_id": 2,
|
| 5 |
+
"pad_token_id": 0,
|
| 6 |
+
"transformers_version": "4.28.1"
|
| 7 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/0e9c94dde3f01cda84a45287cabd0c22ab23cbac
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "ai-forever/ruT5-base",
|
| 3 |
+
"_num_labels": 2,
|
| 4 |
+
"architectures": [
|
| 5 |
+
"T5ForConditionalGeneration"
|
| 6 |
+
],
|
| 7 |
+
"d_ff": 3072,
|
| 8 |
+
"d_kv": 64,
|
| 9 |
+
"d_model": 768,
|
| 10 |
+
"decoder_start_token_id": 0,
|
| 11 |
+
"dense_act_fn": "relu",
|
| 12 |
+
"dropout_rate": 0.1,
|
| 13 |
+
"eos_token_id": 2,
|
| 14 |
+
"feed_forward_proj": "relu",
|
| 15 |
+
"initializer_factor": 1.0,
|
| 16 |
+
"is_encoder_decoder": true,
|
| 17 |
+
"is_gated_act": false,
|
| 18 |
+
"layer_norm_epsilon": 1e-06,
|
| 19 |
+
"model_type": "t5",
|
| 20 |
+
"n_positions": 512,
|
| 21 |
+
"num_decoder_layers": 12,
|
| 22 |
+
"num_heads": 12,
|
| 23 |
+
"num_layers": 12,
|
| 24 |
+
"output_past": true,
|
| 25 |
+
"pad_token_id": 0,
|
| 26 |
+
"relative_attention_max_distance": 128,
|
| 27 |
+
"relative_attention_num_buckets": 32,
|
| 28 |
+
"task_specific_params": {
|
| 29 |
+
"summarization": {
|
| 30 |
+
"early_stopping": true,
|
| 31 |
+
"length_penalty": 2.0,
|
| 32 |
+
"max_length": 200,
|
| 33 |
+
"min_length": 30,
|
| 34 |
+
"no_repeat_ngram_size": 3,
|
| 35 |
+
"num_beams": 4,
|
| 36 |
+
"prefix": "summarize: "
|
| 37 |
+
},
|
| 38 |
+
"translation_en_to_de": {
|
| 39 |
+
"early_stopping": true,
|
| 40 |
+
"max_length": 300,
|
| 41 |
+
"num_beams": 4,
|
| 42 |
+
"prefix": "translate English to German: "
|
| 43 |
+
},
|
| 44 |
+
"translation_en_to_fr": {
|
| 45 |
+
"early_stopping": true,
|
| 46 |
+
"max_length": 300,
|
| 47 |
+
"num_beams": 4,
|
| 48 |
+
"prefix": "translate English to French: "
|
| 49 |
+
},
|
| 50 |
+
"translation_en_to_ro": {
|
| 51 |
+
"early_stopping": true,
|
| 52 |
+
"max_length": 300,
|
| 53 |
+
"num_beams": 4,
|
| 54 |
+
"prefix": "translate English to Romanian: "
|
| 55 |
+
}
|
| 56 |
+
},
|
| 57 |
+
"torch_dtype": "bfloat16",
|
| 58 |
+
"transformers_version": "4.28.1",
|
| 59 |
+
"use_cache": true,
|
| 60 |
+
"vocab_size": 32128
|
| 61 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/12e7774c6c9be70934c0fd405e2fef3905609e14
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/196088ae0adbafc572d99e62d0090a4a25a0bb305493ff454a7f9fb2171c1566.incomplete
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2f7b2d909e6eaf1d4a9be43e286fb16693d66430ce662b19351b8662539497bb
|
| 3 |
+
size 178257920
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/4fe7c3c6498b6c26f0b5b6c60265b4d88ab4db02e1bbfb7f06cea3dc4746874b
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4fe7c3c6498b6c26f0b5b6c60265b4d88ab4db02e1bbfb7f06cea3dc4746874b
|
| 3 |
+
size 445895825
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/5540ad7d1f6add43e0550aed386b556455404faa
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"additional_special_tokens": [
|
| 3 |
+
"<extra_id_0>",
|
| 4 |
+
"<extra_id_1>",
|
| 5 |
+
"<extra_id_2>",
|
| 6 |
+
"<extra_id_3>",
|
| 7 |
+
"<extra_id_4>",
|
| 8 |
+
"<extra_id_5>",
|
| 9 |
+
"<extra_id_6>",
|
| 10 |
+
"<extra_id_7>",
|
| 11 |
+
"<extra_id_8>",
|
| 12 |
+
"<extra_id_9>",
|
| 13 |
+
"<extra_id_10>",
|
| 14 |
+
"<extra_id_11>",
|
| 15 |
+
"<extra_id_12>",
|
| 16 |
+
"<extra_id_13>",
|
| 17 |
+
"<extra_id_14>",
|
| 18 |
+
"<extra_id_15>",
|
| 19 |
+
"<extra_id_16>",
|
| 20 |
+
"<extra_id_17>",
|
| 21 |
+
"<extra_id_18>",
|
| 22 |
+
"<extra_id_19>",
|
| 23 |
+
"<extra_id_20>",
|
| 24 |
+
"<extra_id_21>",
|
| 25 |
+
"<extra_id_22>",
|
| 26 |
+
"<extra_id_23>",
|
| 27 |
+
"<extra_id_24>",
|
| 28 |
+
"<extra_id_25>",
|
| 29 |
+
"<extra_id_26>",
|
| 30 |
+
"<extra_id_27>",
|
| 31 |
+
"<extra_id_28>",
|
| 32 |
+
"<extra_id_29>",
|
| 33 |
+
"<extra_id_30>",
|
| 34 |
+
"<extra_id_31>",
|
| 35 |
+
"<extra_id_32>",
|
| 36 |
+
"<extra_id_33>",
|
| 37 |
+
"<extra_id_34>",
|
| 38 |
+
"<extra_id_35>",
|
| 39 |
+
"<extra_id_36>",
|
| 40 |
+
"<extra_id_37>",
|
| 41 |
+
"<extra_id_38>",
|
| 42 |
+
"<extra_id_39>",
|
| 43 |
+
"<extra_id_40>",
|
| 44 |
+
"<extra_id_41>",
|
| 45 |
+
"<extra_id_42>",
|
| 46 |
+
"<extra_id_43>",
|
| 47 |
+
"<extra_id_44>",
|
| 48 |
+
"<extra_id_45>",
|
| 49 |
+
"<extra_id_46>",
|
| 50 |
+
"<extra_id_47>",
|
| 51 |
+
"<extra_id_48>",
|
| 52 |
+
"<extra_id_49>",
|
| 53 |
+
"<extra_id_50>",
|
| 54 |
+
"<extra_id_51>",
|
| 55 |
+
"<extra_id_52>",
|
| 56 |
+
"<extra_id_53>",
|
| 57 |
+
"<extra_id_54>",
|
| 58 |
+
"<extra_id_55>",
|
| 59 |
+
"<extra_id_56>",
|
| 60 |
+
"<extra_id_57>",
|
| 61 |
+
"<extra_id_58>",
|
| 62 |
+
"<extra_id_59>",
|
| 63 |
+
"<extra_id_60>",
|
| 64 |
+
"<extra_id_61>",
|
| 65 |
+
"<extra_id_62>",
|
| 66 |
+
"<extra_id_63>",
|
| 67 |
+
"<extra_id_64>",
|
| 68 |
+
"<extra_id_65>",
|
| 69 |
+
"<extra_id_66>",
|
| 70 |
+
"<extra_id_67>",
|
| 71 |
+
"<extra_id_68>",
|
| 72 |
+
"<extra_id_69>",
|
| 73 |
+
"<extra_id_70>",
|
| 74 |
+
"<extra_id_71>",
|
| 75 |
+
"<extra_id_72>",
|
| 76 |
+
"<extra_id_73>",
|
| 77 |
+
"<extra_id_74>",
|
| 78 |
+
"<extra_id_75>",
|
| 79 |
+
"<extra_id_76>",
|
| 80 |
+
"<extra_id_77>",
|
| 81 |
+
"<extra_id_78>",
|
| 82 |
+
"<extra_id_79>",
|
| 83 |
+
"<extra_id_80>",
|
| 84 |
+
"<extra_id_81>",
|
| 85 |
+
"<extra_id_82>",
|
| 86 |
+
"<extra_id_83>",
|
| 87 |
+
"<extra_id_84>",
|
| 88 |
+
"<extra_id_85>",
|
| 89 |
+
"<extra_id_86>",
|
| 90 |
+
"<extra_id_87>",
|
| 91 |
+
"<extra_id_88>",
|
| 92 |
+
"<extra_id_89>",
|
| 93 |
+
"<extra_id_90>",
|
| 94 |
+
"<extra_id_91>",
|
| 95 |
+
"<extra_id_92>",
|
| 96 |
+
"<extra_id_93>",
|
| 97 |
+
"<extra_id_94>",
|
| 98 |
+
"<extra_id_95>",
|
| 99 |
+
"<extra_id_96>",
|
| 100 |
+
"<extra_id_97>",
|
| 101 |
+
"<extra_id_98>",
|
| 102 |
+
"<extra_id_99>"
|
| 103 |
+
],
|
| 104 |
+
"bos_token": "<s>",
|
| 105 |
+
"eos_token": "</s>",
|
| 106 |
+
"pad_token": "<pad>",
|
| 107 |
+
"unk_token": "<unk>"
|
| 108 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/5750775dfac6c13088a6c54dd07cf88f7f9cee1d
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"<s>": 32100
|
| 3 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/7a4eb87011448a4564a3144979384da51eee1da95e554feb22ccc85529535dd5
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7a4eb87011448a4564a3144979384da51eee1da95e554feb22ccc85529535dd5
|
| 3 |
+
size 1003118
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/blobs/94fe40201a0d25ffe6e448ae1a91f41fc67d28fa
ADDED
|
@@ -0,0 +1,938 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"added_tokens_decoder": {
|
| 3 |
+
"0": {
|
| 4 |
+
"content": "<pad>",
|
| 5 |
+
"lstrip": false,
|
| 6 |
+
"normalized": false,
|
| 7 |
+
"rstrip": false,
|
| 8 |
+
"single_word": false,
|
| 9 |
+
"special": true
|
| 10 |
+
},
|
| 11 |
+
"1": {
|
| 12 |
+
"content": "<unk>",
|
| 13 |
+
"lstrip": false,
|
| 14 |
+
"normalized": false,
|
| 15 |
+
"rstrip": false,
|
| 16 |
+
"single_word": false,
|
| 17 |
+
"special": true
|
| 18 |
+
},
|
| 19 |
+
"2": {
|
| 20 |
+
"content": "</s>",
|
| 21 |
+
"lstrip": false,
|
| 22 |
+
"normalized": false,
|
| 23 |
+
"rstrip": false,
|
| 24 |
+
"single_word": false,
|
| 25 |
+
"special": true
|
| 26 |
+
},
|
| 27 |
+
"32000": {
|
| 28 |
+
"content": "<extra_id_99>",
|
| 29 |
+
"lstrip": true,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": true,
|
| 32 |
+
"single_word": true,
|
| 33 |
+
"special": true
|
| 34 |
+
},
|
| 35 |
+
"32001": {
|
| 36 |
+
"content": "<extra_id_98>",
|
| 37 |
+
"lstrip": true,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": true,
|
| 40 |
+
"single_word": true,
|
| 41 |
+
"special": true
|
| 42 |
+
},
|
| 43 |
+
"32002": {
|
| 44 |
+
"content": "<extra_id_97>",
|
| 45 |
+
"lstrip": true,
|
| 46 |
+
"normalized": false,
|
| 47 |
+
"rstrip": true,
|
| 48 |
+
"single_word": true,
|
| 49 |
+
"special": true
|
| 50 |
+
},
|
| 51 |
+
"32003": {
|
| 52 |
+
"content": "<extra_id_96>",
|
| 53 |
+
"lstrip": true,
|
| 54 |
+
"normalized": false,
|
| 55 |
+
"rstrip": true,
|
| 56 |
+
"single_word": true,
|
| 57 |
+
"special": true
|
| 58 |
+
},
|
| 59 |
+
"32004": {
|
| 60 |
+
"content": "<extra_id_95>",
|
| 61 |
+
"lstrip": true,
|
| 62 |
+
"normalized": false,
|
| 63 |
+
"rstrip": true,
|
| 64 |
+
"single_word": true,
|
| 65 |
+
"special": true
|
| 66 |
+
},
|
| 67 |
+
"32005": {
|
| 68 |
+
"content": "<extra_id_94>",
|
| 69 |
+
"lstrip": true,
|
| 70 |
+
"normalized": false,
|
| 71 |
+
"rstrip": true,
|
| 72 |
+
"single_word": true,
|
| 73 |
+
"special": true
|
| 74 |
+
},
|
| 75 |
+
"32006": {
|
| 76 |
+
"content": "<extra_id_93>",
|
| 77 |
+
"lstrip": true,
|
| 78 |
+
"normalized": false,
|
| 79 |
+
"rstrip": true,
|
| 80 |
+
"single_word": true,
|
| 81 |
+
"special": true
|
| 82 |
+
},
|
| 83 |
+
"32007": {
|
| 84 |
+
"content": "<extra_id_92>",
|
| 85 |
+
"lstrip": true,
|
| 86 |
+
"normalized": false,
|
| 87 |
+
"rstrip": true,
|
| 88 |
+
"single_word": true,
|
| 89 |
+
"special": true
|
| 90 |
+
},
|
| 91 |
+
"32008": {
|
| 92 |
+
"content": "<extra_id_91>",
|
| 93 |
+
"lstrip": true,
|
| 94 |
+
"normalized": false,
|
| 95 |
+
"rstrip": true,
|
| 96 |
+
"single_word": true,
|
| 97 |
+
"special": true
|
| 98 |
+
},
|
| 99 |
+
"32009": {
|
| 100 |
+
"content": "<extra_id_90>",
|
| 101 |
+
"lstrip": true,
|
| 102 |
+
"normalized": false,
|
| 103 |
+
"rstrip": true,
|
| 104 |
+
"single_word": true,
|
| 105 |
+
"special": true
|
| 106 |
+
},
|
| 107 |
+
"32010": {
|
| 108 |
+
"content": "<extra_id_89>",
|
| 109 |
+
"lstrip": true,
|
| 110 |
+
"normalized": false,
|
| 111 |
+
"rstrip": true,
|
| 112 |
+
"single_word": true,
|
| 113 |
+
"special": true
|
| 114 |
+
},
|
| 115 |
+
"32011": {
|
| 116 |
+
"content": "<extra_id_88>",
|
| 117 |
+
"lstrip": true,
|
| 118 |
+
"normalized": false,
|
| 119 |
+
"rstrip": true,
|
| 120 |
+
"single_word": true,
|
| 121 |
+
"special": true
|
| 122 |
+
},
|
| 123 |
+
"32012": {
|
| 124 |
+
"content": "<extra_id_87>",
|
| 125 |
+
"lstrip": true,
|
| 126 |
+
"normalized": false,
|
| 127 |
+
"rstrip": true,
|
| 128 |
+
"single_word": true,
|
| 129 |
+
"special": true
|
| 130 |
+
},
|
| 131 |
+
"32013": {
|
| 132 |
+
"content": "<extra_id_86>",
|
| 133 |
+
"lstrip": true,
|
| 134 |
+
"normalized": false,
|
| 135 |
+
"rstrip": true,
|
| 136 |
+
"single_word": true,
|
| 137 |
+
"special": true
|
| 138 |
+
},
|
| 139 |
+
"32014": {
|
| 140 |
+
"content": "<extra_id_85>",
|
| 141 |
+
"lstrip": true,
|
| 142 |
+
"normalized": false,
|
| 143 |
+
"rstrip": true,
|
| 144 |
+
"single_word": true,
|
| 145 |
+
"special": true
|
| 146 |
+
},
|
| 147 |
+
"32015": {
|
| 148 |
+
"content": "<extra_id_84>",
|
| 149 |
+
"lstrip": true,
|
| 150 |
+
"normalized": false,
|
| 151 |
+
"rstrip": true,
|
| 152 |
+
"single_word": true,
|
| 153 |
+
"special": true
|
| 154 |
+
},
|
| 155 |
+
"32016": {
|
| 156 |
+
"content": "<extra_id_83>",
|
| 157 |
+
"lstrip": true,
|
| 158 |
+
"normalized": false,
|
| 159 |
+
"rstrip": true,
|
| 160 |
+
"single_word": true,
|
| 161 |
+
"special": true
|
| 162 |
+
},
|
| 163 |
+
"32017": {
|
| 164 |
+
"content": "<extra_id_82>",
|
| 165 |
+
"lstrip": true,
|
| 166 |
+
"normalized": false,
|
| 167 |
+
"rstrip": true,
|
| 168 |
+
"single_word": true,
|
| 169 |
+
"special": true
|
| 170 |
+
},
|
| 171 |
+
"32018": {
|
| 172 |
+
"content": "<extra_id_81>",
|
| 173 |
+
"lstrip": true,
|
| 174 |
+
"normalized": false,
|
| 175 |
+
"rstrip": true,
|
| 176 |
+
"single_word": true,
|
| 177 |
+
"special": true
|
| 178 |
+
},
|
| 179 |
+
"32019": {
|
| 180 |
+
"content": "<extra_id_80>",
|
| 181 |
+
"lstrip": true,
|
| 182 |
+
"normalized": false,
|
| 183 |
+
"rstrip": true,
|
| 184 |
+
"single_word": true,
|
| 185 |
+
"special": true
|
| 186 |
+
},
|
| 187 |
+
"32020": {
|
| 188 |
+
"content": "<extra_id_79>",
|
| 189 |
+
"lstrip": true,
|
| 190 |
+
"normalized": false,
|
| 191 |
+
"rstrip": true,
|
| 192 |
+
"single_word": true,
|
| 193 |
+
"special": true
|
| 194 |
+
},
|
| 195 |
+
"32021": {
|
| 196 |
+
"content": "<extra_id_78>",
|
| 197 |
+
"lstrip": true,
|
| 198 |
+
"normalized": false,
|
| 199 |
+
"rstrip": true,
|
| 200 |
+
"single_word": true,
|
| 201 |
+
"special": true
|
| 202 |
+
},
|
| 203 |
+
"32022": {
|
| 204 |
+
"content": "<extra_id_77>",
|
| 205 |
+
"lstrip": true,
|
| 206 |
+
"normalized": false,
|
| 207 |
+
"rstrip": true,
|
| 208 |
+
"single_word": true,
|
| 209 |
+
"special": true
|
| 210 |
+
},
|
| 211 |
+
"32023": {
|
| 212 |
+
"content": "<extra_id_76>",
|
| 213 |
+
"lstrip": true,
|
| 214 |
+
"normalized": false,
|
| 215 |
+
"rstrip": true,
|
| 216 |
+
"single_word": true,
|
| 217 |
+
"special": true
|
| 218 |
+
},
|
| 219 |
+
"32024": {
|
| 220 |
+
"content": "<extra_id_75>",
|
| 221 |
+
"lstrip": true,
|
| 222 |
+
"normalized": false,
|
| 223 |
+
"rstrip": true,
|
| 224 |
+
"single_word": true,
|
| 225 |
+
"special": true
|
| 226 |
+
},
|
| 227 |
+
"32025": {
|
| 228 |
+
"content": "<extra_id_74>",
|
| 229 |
+
"lstrip": true,
|
| 230 |
+
"normalized": false,
|
| 231 |
+
"rstrip": true,
|
| 232 |
+
"single_word": true,
|
| 233 |
+
"special": true
|
| 234 |
+
},
|
| 235 |
+
"32026": {
|
| 236 |
+
"content": "<extra_id_73>",
|
| 237 |
+
"lstrip": true,
|
| 238 |
+
"normalized": false,
|
| 239 |
+
"rstrip": true,
|
| 240 |
+
"single_word": true,
|
| 241 |
+
"special": true
|
| 242 |
+
},
|
| 243 |
+
"32027": {
|
| 244 |
+
"content": "<extra_id_72>",
|
| 245 |
+
"lstrip": true,
|
| 246 |
+
"normalized": false,
|
| 247 |
+
"rstrip": true,
|
| 248 |
+
"single_word": true,
|
| 249 |
+
"special": true
|
| 250 |
+
},
|
| 251 |
+
"32028": {
|
| 252 |
+
"content": "<extra_id_71>",
|
| 253 |
+
"lstrip": true,
|
| 254 |
+
"normalized": false,
|
| 255 |
+
"rstrip": true,
|
| 256 |
+
"single_word": true,
|
| 257 |
+
"special": true
|
| 258 |
+
},
|
| 259 |
+
"32029": {
|
| 260 |
+
"content": "<extra_id_70>",
|
| 261 |
+
"lstrip": true,
|
| 262 |
+
"normalized": false,
|
| 263 |
+
"rstrip": true,
|
| 264 |
+
"single_word": true,
|
| 265 |
+
"special": true
|
| 266 |
+
},
|
| 267 |
+
"32030": {
|
| 268 |
+
"content": "<extra_id_69>",
|
| 269 |
+
"lstrip": true,
|
| 270 |
+
"normalized": false,
|
| 271 |
+
"rstrip": true,
|
| 272 |
+
"single_word": true,
|
| 273 |
+
"special": true
|
| 274 |
+
},
|
| 275 |
+
"32031": {
|
| 276 |
+
"content": "<extra_id_68>",
|
| 277 |
+
"lstrip": true,
|
| 278 |
+
"normalized": false,
|
| 279 |
+
"rstrip": true,
|
| 280 |
+
"single_word": true,
|
| 281 |
+
"special": true
|
| 282 |
+
},
|
| 283 |
+
"32032": {
|
| 284 |
+
"content": "<extra_id_67>",
|
| 285 |
+
"lstrip": true,
|
| 286 |
+
"normalized": false,
|
| 287 |
+
"rstrip": true,
|
| 288 |
+
"single_word": true,
|
| 289 |
+
"special": true
|
| 290 |
+
},
|
| 291 |
+
"32033": {
|
| 292 |
+
"content": "<extra_id_66>",
|
| 293 |
+
"lstrip": true,
|
| 294 |
+
"normalized": false,
|
| 295 |
+
"rstrip": true,
|
| 296 |
+
"single_word": true,
|
| 297 |
+
"special": true
|
| 298 |
+
},
|
| 299 |
+
"32034": {
|
| 300 |
+
"content": "<extra_id_65>",
|
| 301 |
+
"lstrip": true,
|
| 302 |
+
"normalized": false,
|
| 303 |
+
"rstrip": true,
|
| 304 |
+
"single_word": true,
|
| 305 |
+
"special": true
|
| 306 |
+
},
|
| 307 |
+
"32035": {
|
| 308 |
+
"content": "<extra_id_64>",
|
| 309 |
+
"lstrip": true,
|
| 310 |
+
"normalized": false,
|
| 311 |
+
"rstrip": true,
|
| 312 |
+
"single_word": true,
|
| 313 |
+
"special": true
|
| 314 |
+
},
|
| 315 |
+
"32036": {
|
| 316 |
+
"content": "<extra_id_63>",
|
| 317 |
+
"lstrip": true,
|
| 318 |
+
"normalized": false,
|
| 319 |
+
"rstrip": true,
|
| 320 |
+
"single_word": true,
|
| 321 |
+
"special": true
|
| 322 |
+
},
|
| 323 |
+
"32037": {
|
| 324 |
+
"content": "<extra_id_62>",
|
| 325 |
+
"lstrip": true,
|
| 326 |
+
"normalized": false,
|
| 327 |
+
"rstrip": true,
|
| 328 |
+
"single_word": true,
|
| 329 |
+
"special": true
|
| 330 |
+
},
|
| 331 |
+
"32038": {
|
| 332 |
+
"content": "<extra_id_61>",
|
| 333 |
+
"lstrip": true,
|
| 334 |
+
"normalized": false,
|
| 335 |
+
"rstrip": true,
|
| 336 |
+
"single_word": true,
|
| 337 |
+
"special": true
|
| 338 |
+
},
|
| 339 |
+
"32039": {
|
| 340 |
+
"content": "<extra_id_60>",
|
| 341 |
+
"lstrip": true,
|
| 342 |
+
"normalized": false,
|
| 343 |
+
"rstrip": true,
|
| 344 |
+
"single_word": true,
|
| 345 |
+
"special": true
|
| 346 |
+
},
|
| 347 |
+
"32040": {
|
| 348 |
+
"content": "<extra_id_59>",
|
| 349 |
+
"lstrip": true,
|
| 350 |
+
"normalized": false,
|
| 351 |
+
"rstrip": true,
|
| 352 |
+
"single_word": true,
|
| 353 |
+
"special": true
|
| 354 |
+
},
|
| 355 |
+
"32041": {
|
| 356 |
+
"content": "<extra_id_58>",
|
| 357 |
+
"lstrip": true,
|
| 358 |
+
"normalized": false,
|
| 359 |
+
"rstrip": true,
|
| 360 |
+
"single_word": true,
|
| 361 |
+
"special": true
|
| 362 |
+
},
|
| 363 |
+
"32042": {
|
| 364 |
+
"content": "<extra_id_57>",
|
| 365 |
+
"lstrip": true,
|
| 366 |
+
"normalized": false,
|
| 367 |
+
"rstrip": true,
|
| 368 |
+
"single_word": true,
|
| 369 |
+
"special": true
|
| 370 |
+
},
|
| 371 |
+
"32043": {
|
| 372 |
+
"content": "<extra_id_56>",
|
| 373 |
+
"lstrip": true,
|
| 374 |
+
"normalized": false,
|
| 375 |
+
"rstrip": true,
|
| 376 |
+
"single_word": true,
|
| 377 |
+
"special": true
|
| 378 |
+
},
|
| 379 |
+
"32044": {
|
| 380 |
+
"content": "<extra_id_55>",
|
| 381 |
+
"lstrip": true,
|
| 382 |
+
"normalized": false,
|
| 383 |
+
"rstrip": true,
|
| 384 |
+
"single_word": true,
|
| 385 |
+
"special": true
|
| 386 |
+
},
|
| 387 |
+
"32045": {
|
| 388 |
+
"content": "<extra_id_54>",
|
| 389 |
+
"lstrip": true,
|
| 390 |
+
"normalized": false,
|
| 391 |
+
"rstrip": true,
|
| 392 |
+
"single_word": true,
|
| 393 |
+
"special": true
|
| 394 |
+
},
|
| 395 |
+
"32046": {
|
| 396 |
+
"content": "<extra_id_53>",
|
| 397 |
+
"lstrip": true,
|
| 398 |
+
"normalized": false,
|
| 399 |
+
"rstrip": true,
|
| 400 |
+
"single_word": true,
|
| 401 |
+
"special": true
|
| 402 |
+
},
|
| 403 |
+
"32047": {
|
| 404 |
+
"content": "<extra_id_52>",
|
| 405 |
+
"lstrip": true,
|
| 406 |
+
"normalized": false,
|
| 407 |
+
"rstrip": true,
|
| 408 |
+
"single_word": true,
|
| 409 |
+
"special": true
|
| 410 |
+
},
|
| 411 |
+
"32048": {
|
| 412 |
+
"content": "<extra_id_51>",
|
| 413 |
+
"lstrip": true,
|
| 414 |
+
"normalized": false,
|
| 415 |
+
"rstrip": true,
|
| 416 |
+
"single_word": true,
|
| 417 |
+
"special": true
|
| 418 |
+
},
|
| 419 |
+
"32049": {
|
| 420 |
+
"content": "<extra_id_50>",
|
| 421 |
+
"lstrip": true,
|
| 422 |
+
"normalized": false,
|
| 423 |
+
"rstrip": true,
|
| 424 |
+
"single_word": true,
|
| 425 |
+
"special": true
|
| 426 |
+
},
|
| 427 |
+
"32050": {
|
| 428 |
+
"content": "<extra_id_49>",
|
| 429 |
+
"lstrip": true,
|
| 430 |
+
"normalized": false,
|
| 431 |
+
"rstrip": true,
|
| 432 |
+
"single_word": true,
|
| 433 |
+
"special": true
|
| 434 |
+
},
|
| 435 |
+
"32051": {
|
| 436 |
+
"content": "<extra_id_48>",
|
| 437 |
+
"lstrip": true,
|
| 438 |
+
"normalized": false,
|
| 439 |
+
"rstrip": true,
|
| 440 |
+
"single_word": true,
|
| 441 |
+
"special": true
|
| 442 |
+
},
|
| 443 |
+
"32052": {
|
| 444 |
+
"content": "<extra_id_47>",
|
| 445 |
+
"lstrip": true,
|
| 446 |
+
"normalized": false,
|
| 447 |
+
"rstrip": true,
|
| 448 |
+
"single_word": true,
|
| 449 |
+
"special": true
|
| 450 |
+
},
|
| 451 |
+
"32053": {
|
| 452 |
+
"content": "<extra_id_46>",
|
| 453 |
+
"lstrip": true,
|
| 454 |
+
"normalized": false,
|
| 455 |
+
"rstrip": true,
|
| 456 |
+
"single_word": true,
|
| 457 |
+
"special": true
|
| 458 |
+
},
|
| 459 |
+
"32054": {
|
| 460 |
+
"content": "<extra_id_45>",
|
| 461 |
+
"lstrip": true,
|
| 462 |
+
"normalized": false,
|
| 463 |
+
"rstrip": true,
|
| 464 |
+
"single_word": true,
|
| 465 |
+
"special": true
|
| 466 |
+
},
|
| 467 |
+
"32055": {
|
| 468 |
+
"content": "<extra_id_44>",
|
| 469 |
+
"lstrip": true,
|
| 470 |
+
"normalized": false,
|
| 471 |
+
"rstrip": true,
|
| 472 |
+
"single_word": true,
|
| 473 |
+
"special": true
|
| 474 |
+
},
|
| 475 |
+
"32056": {
|
| 476 |
+
"content": "<extra_id_43>",
|
| 477 |
+
"lstrip": true,
|
| 478 |
+
"normalized": false,
|
| 479 |
+
"rstrip": true,
|
| 480 |
+
"single_word": true,
|
| 481 |
+
"special": true
|
| 482 |
+
},
|
| 483 |
+
"32057": {
|
| 484 |
+
"content": "<extra_id_42>",
|
| 485 |
+
"lstrip": true,
|
| 486 |
+
"normalized": false,
|
| 487 |
+
"rstrip": true,
|
| 488 |
+
"single_word": true,
|
| 489 |
+
"special": true
|
| 490 |
+
},
|
| 491 |
+
"32058": {
|
| 492 |
+
"content": "<extra_id_41>",
|
| 493 |
+
"lstrip": true,
|
| 494 |
+
"normalized": false,
|
| 495 |
+
"rstrip": true,
|
| 496 |
+
"single_word": true,
|
| 497 |
+
"special": true
|
| 498 |
+
},
|
| 499 |
+
"32059": {
|
| 500 |
+
"content": "<extra_id_40>",
|
| 501 |
+
"lstrip": true,
|
| 502 |
+
"normalized": false,
|
| 503 |
+
"rstrip": true,
|
| 504 |
+
"single_word": true,
|
| 505 |
+
"special": true
|
| 506 |
+
},
|
| 507 |
+
"32060": {
|
| 508 |
+
"content": "<extra_id_39>",
|
| 509 |
+
"lstrip": true,
|
| 510 |
+
"normalized": false,
|
| 511 |
+
"rstrip": true,
|
| 512 |
+
"single_word": true,
|
| 513 |
+
"special": true
|
| 514 |
+
},
|
| 515 |
+
"32061": {
|
| 516 |
+
"content": "<extra_id_38>",
|
| 517 |
+
"lstrip": true,
|
| 518 |
+
"normalized": false,
|
| 519 |
+
"rstrip": true,
|
| 520 |
+
"single_word": true,
|
| 521 |
+
"special": true
|
| 522 |
+
},
|
| 523 |
+
"32062": {
|
| 524 |
+
"content": "<extra_id_37>",
|
| 525 |
+
"lstrip": true,
|
| 526 |
+
"normalized": false,
|
| 527 |
+
"rstrip": true,
|
| 528 |
+
"single_word": true,
|
| 529 |
+
"special": true
|
| 530 |
+
},
|
| 531 |
+
"32063": {
|
| 532 |
+
"content": "<extra_id_36>",
|
| 533 |
+
"lstrip": true,
|
| 534 |
+
"normalized": false,
|
| 535 |
+
"rstrip": true,
|
| 536 |
+
"single_word": true,
|
| 537 |
+
"special": true
|
| 538 |
+
},
|
| 539 |
+
"32064": {
|
| 540 |
+
"content": "<extra_id_35>",
|
| 541 |
+
"lstrip": true,
|
| 542 |
+
"normalized": false,
|
| 543 |
+
"rstrip": true,
|
| 544 |
+
"single_word": true,
|
| 545 |
+
"special": true
|
| 546 |
+
},
|
| 547 |
+
"32065": {
|
| 548 |
+
"content": "<extra_id_34>",
|
| 549 |
+
"lstrip": true,
|
| 550 |
+
"normalized": false,
|
| 551 |
+
"rstrip": true,
|
| 552 |
+
"single_word": true,
|
| 553 |
+
"special": true
|
| 554 |
+
},
|
| 555 |
+
"32066": {
|
| 556 |
+
"content": "<extra_id_33>",
|
| 557 |
+
"lstrip": true,
|
| 558 |
+
"normalized": false,
|
| 559 |
+
"rstrip": true,
|
| 560 |
+
"single_word": true,
|
| 561 |
+
"special": true
|
| 562 |
+
},
|
| 563 |
+
"32067": {
|
| 564 |
+
"content": "<extra_id_32>",
|
| 565 |
+
"lstrip": true,
|
| 566 |
+
"normalized": false,
|
| 567 |
+
"rstrip": true,
|
| 568 |
+
"single_word": true,
|
| 569 |
+
"special": true
|
| 570 |
+
},
|
| 571 |
+
"32068": {
|
| 572 |
+
"content": "<extra_id_31>",
|
| 573 |
+
"lstrip": true,
|
| 574 |
+
"normalized": false,
|
| 575 |
+
"rstrip": true,
|
| 576 |
+
"single_word": true,
|
| 577 |
+
"special": true
|
| 578 |
+
},
|
| 579 |
+
"32069": {
|
| 580 |
+
"content": "<extra_id_30>",
|
| 581 |
+
"lstrip": true,
|
| 582 |
+
"normalized": false,
|
| 583 |
+
"rstrip": true,
|
| 584 |
+
"single_word": true,
|
| 585 |
+
"special": true
|
| 586 |
+
},
|
| 587 |
+
"32070": {
|
| 588 |
+
"content": "<extra_id_29>",
|
| 589 |
+
"lstrip": true,
|
| 590 |
+
"normalized": false,
|
| 591 |
+
"rstrip": true,
|
| 592 |
+
"single_word": true,
|
| 593 |
+
"special": true
|
| 594 |
+
},
|
| 595 |
+
"32071": {
|
| 596 |
+
"content": "<extra_id_28>",
|
| 597 |
+
"lstrip": true,
|
| 598 |
+
"normalized": false,
|
| 599 |
+
"rstrip": true,
|
| 600 |
+
"single_word": true,
|
| 601 |
+
"special": true
|
| 602 |
+
},
|
| 603 |
+
"32072": {
|
| 604 |
+
"content": "<extra_id_27>",
|
| 605 |
+
"lstrip": true,
|
| 606 |
+
"normalized": false,
|
| 607 |
+
"rstrip": true,
|
| 608 |
+
"single_word": true,
|
| 609 |
+
"special": true
|
| 610 |
+
},
|
| 611 |
+
"32073": {
|
| 612 |
+
"content": "<extra_id_26>",
|
| 613 |
+
"lstrip": true,
|
| 614 |
+
"normalized": false,
|
| 615 |
+
"rstrip": true,
|
| 616 |
+
"single_word": true,
|
| 617 |
+
"special": true
|
| 618 |
+
},
|
| 619 |
+
"32074": {
|
| 620 |
+
"content": "<extra_id_25>",
|
| 621 |
+
"lstrip": true,
|
| 622 |
+
"normalized": false,
|
| 623 |
+
"rstrip": true,
|
| 624 |
+
"single_word": true,
|
| 625 |
+
"special": true
|
| 626 |
+
},
|
| 627 |
+
"32075": {
|
| 628 |
+
"content": "<extra_id_24>",
|
| 629 |
+
"lstrip": true,
|
| 630 |
+
"normalized": false,
|
| 631 |
+
"rstrip": true,
|
| 632 |
+
"single_word": true,
|
| 633 |
+
"special": true
|
| 634 |
+
},
|
| 635 |
+
"32076": {
|
| 636 |
+
"content": "<extra_id_23>",
|
| 637 |
+
"lstrip": true,
|
| 638 |
+
"normalized": false,
|
| 639 |
+
"rstrip": true,
|
| 640 |
+
"single_word": true,
|
| 641 |
+
"special": true
|
| 642 |
+
},
|
| 643 |
+
"32077": {
|
| 644 |
+
"content": "<extra_id_22>",
|
| 645 |
+
"lstrip": true,
|
| 646 |
+
"normalized": false,
|
| 647 |
+
"rstrip": true,
|
| 648 |
+
"single_word": true,
|
| 649 |
+
"special": true
|
| 650 |
+
},
|
| 651 |
+
"32078": {
|
| 652 |
+
"content": "<extra_id_21>",
|
| 653 |
+
"lstrip": true,
|
| 654 |
+
"normalized": false,
|
| 655 |
+
"rstrip": true,
|
| 656 |
+
"single_word": true,
|
| 657 |
+
"special": true
|
| 658 |
+
},
|
| 659 |
+
"32079": {
|
| 660 |
+
"content": "<extra_id_20>",
|
| 661 |
+
"lstrip": true,
|
| 662 |
+
"normalized": false,
|
| 663 |
+
"rstrip": true,
|
| 664 |
+
"single_word": true,
|
| 665 |
+
"special": true
|
| 666 |
+
},
|
| 667 |
+
"32080": {
|
| 668 |
+
"content": "<extra_id_19>",
|
| 669 |
+
"lstrip": true,
|
| 670 |
+
"normalized": false,
|
| 671 |
+
"rstrip": true,
|
| 672 |
+
"single_word": true,
|
| 673 |
+
"special": true
|
| 674 |
+
},
|
| 675 |
+
"32081": {
|
| 676 |
+
"content": "<extra_id_18>",
|
| 677 |
+
"lstrip": true,
|
| 678 |
+
"normalized": false,
|
| 679 |
+
"rstrip": true,
|
| 680 |
+
"single_word": true,
|
| 681 |
+
"special": true
|
| 682 |
+
},
|
| 683 |
+
"32082": {
|
| 684 |
+
"content": "<extra_id_17>",
|
| 685 |
+
"lstrip": true,
|
| 686 |
+
"normalized": false,
|
| 687 |
+
"rstrip": true,
|
| 688 |
+
"single_word": true,
|
| 689 |
+
"special": true
|
| 690 |
+
},
|
| 691 |
+
"32083": {
|
| 692 |
+
"content": "<extra_id_16>",
|
| 693 |
+
"lstrip": true,
|
| 694 |
+
"normalized": false,
|
| 695 |
+
"rstrip": true,
|
| 696 |
+
"single_word": true,
|
| 697 |
+
"special": true
|
| 698 |
+
},
|
| 699 |
+
"32084": {
|
| 700 |
+
"content": "<extra_id_15>",
|
| 701 |
+
"lstrip": true,
|
| 702 |
+
"normalized": false,
|
| 703 |
+
"rstrip": true,
|
| 704 |
+
"single_word": true,
|
| 705 |
+
"special": true
|
| 706 |
+
},
|
| 707 |
+
"32085": {
|
| 708 |
+
"content": "<extra_id_14>",
|
| 709 |
+
"lstrip": true,
|
| 710 |
+
"normalized": false,
|
| 711 |
+
"rstrip": true,
|
| 712 |
+
"single_word": true,
|
| 713 |
+
"special": true
|
| 714 |
+
},
|
| 715 |
+
"32086": {
|
| 716 |
+
"content": "<extra_id_13>",
|
| 717 |
+
"lstrip": true,
|
| 718 |
+
"normalized": false,
|
| 719 |
+
"rstrip": true,
|
| 720 |
+
"single_word": true,
|
| 721 |
+
"special": true
|
| 722 |
+
},
|
| 723 |
+
"32087": {
|
| 724 |
+
"content": "<extra_id_12>",
|
| 725 |
+
"lstrip": true,
|
| 726 |
+
"normalized": false,
|
| 727 |
+
"rstrip": true,
|
| 728 |
+
"single_word": true,
|
| 729 |
+
"special": true
|
| 730 |
+
},
|
| 731 |
+
"32088": {
|
| 732 |
+
"content": "<extra_id_11>",
|
| 733 |
+
"lstrip": true,
|
| 734 |
+
"normalized": false,
|
| 735 |
+
"rstrip": true,
|
| 736 |
+
"single_word": true,
|
| 737 |
+
"special": true
|
| 738 |
+
},
|
| 739 |
+
"32089": {
|
| 740 |
+
"content": "<extra_id_10>",
|
| 741 |
+
"lstrip": true,
|
| 742 |
+
"normalized": false,
|
| 743 |
+
"rstrip": true,
|
| 744 |
+
"single_word": true,
|
| 745 |
+
"special": true
|
| 746 |
+
},
|
| 747 |
+
"32090": {
|
| 748 |
+
"content": "<extra_id_9>",
|
| 749 |
+
"lstrip": true,
|
| 750 |
+
"normalized": false,
|
| 751 |
+
"rstrip": true,
|
| 752 |
+
"single_word": true,
|
| 753 |
+
"special": true
|
| 754 |
+
},
|
| 755 |
+
"32091": {
|
| 756 |
+
"content": "<extra_id_8>",
|
| 757 |
+
"lstrip": true,
|
| 758 |
+
"normalized": false,
|
| 759 |
+
"rstrip": true,
|
| 760 |
+
"single_word": true,
|
| 761 |
+
"special": true
|
| 762 |
+
},
|
| 763 |
+
"32092": {
|
| 764 |
+
"content": "<extra_id_7>",
|
| 765 |
+
"lstrip": true,
|
| 766 |
+
"normalized": false,
|
| 767 |
+
"rstrip": true,
|
| 768 |
+
"single_word": true,
|
| 769 |
+
"special": true
|
| 770 |
+
},
|
| 771 |
+
"32093": {
|
| 772 |
+
"content": "<extra_id_6>",
|
| 773 |
+
"lstrip": true,
|
| 774 |
+
"normalized": false,
|
| 775 |
+
"rstrip": true,
|
| 776 |
+
"single_word": true,
|
| 777 |
+
"special": true
|
| 778 |
+
},
|
| 779 |
+
"32094": {
|
| 780 |
+
"content": "<extra_id_5>",
|
| 781 |
+
"lstrip": true,
|
| 782 |
+
"normalized": false,
|
| 783 |
+
"rstrip": true,
|
| 784 |
+
"single_word": true,
|
| 785 |
+
"special": true
|
| 786 |
+
},
|
| 787 |
+
"32095": {
|
| 788 |
+
"content": "<extra_id_4>",
|
| 789 |
+
"lstrip": true,
|
| 790 |
+
"normalized": false,
|
| 791 |
+
"rstrip": true,
|
| 792 |
+
"single_word": true,
|
| 793 |
+
"special": true
|
| 794 |
+
},
|
| 795 |
+
"32096": {
|
| 796 |
+
"content": "<extra_id_3>",
|
| 797 |
+
"lstrip": true,
|
| 798 |
+
"normalized": false,
|
| 799 |
+
"rstrip": true,
|
| 800 |
+
"single_word": true,
|
| 801 |
+
"special": true
|
| 802 |
+
},
|
| 803 |
+
"32097": {
|
| 804 |
+
"content": "<extra_id_2>",
|
| 805 |
+
"lstrip": true,
|
| 806 |
+
"normalized": false,
|
| 807 |
+
"rstrip": true,
|
| 808 |
+
"single_word": true,
|
| 809 |
+
"special": true
|
| 810 |
+
},
|
| 811 |
+
"32098": {
|
| 812 |
+
"content": "<extra_id_1>",
|
| 813 |
+
"lstrip": true,
|
| 814 |
+
"normalized": false,
|
| 815 |
+
"rstrip": true,
|
| 816 |
+
"single_word": true,
|
| 817 |
+
"special": true
|
| 818 |
+
},
|
| 819 |
+
"32099": {
|
| 820 |
+
"content": "<extra_id_0>",
|
| 821 |
+
"lstrip": true,
|
| 822 |
+
"normalized": false,
|
| 823 |
+
"rstrip": true,
|
| 824 |
+
"single_word": true,
|
| 825 |
+
"special": true
|
| 826 |
+
}
|
| 827 |
+
},
|
| 828 |
+
"additional_special_tokens": [
|
| 829 |
+
"<extra_id_0>",
|
| 830 |
+
"<extra_id_1>",
|
| 831 |
+
"<extra_id_2>",
|
| 832 |
+
"<extra_id_3>",
|
| 833 |
+
"<extra_id_4>",
|
| 834 |
+
"<extra_id_5>",
|
| 835 |
+
"<extra_id_6>",
|
| 836 |
+
"<extra_id_7>",
|
| 837 |
+
"<extra_id_8>",
|
| 838 |
+
"<extra_id_9>",
|
| 839 |
+
"<extra_id_10>",
|
| 840 |
+
"<extra_id_11>",
|
| 841 |
+
"<extra_id_12>",
|
| 842 |
+
"<extra_id_13>",
|
| 843 |
+
"<extra_id_14>",
|
| 844 |
+
"<extra_id_15>",
|
| 845 |
+
"<extra_id_16>",
|
| 846 |
+
"<extra_id_17>",
|
| 847 |
+
"<extra_id_18>",
|
| 848 |
+
"<extra_id_19>",
|
| 849 |
+
"<extra_id_20>",
|
| 850 |
+
"<extra_id_21>",
|
| 851 |
+
"<extra_id_22>",
|
| 852 |
+
"<extra_id_23>",
|
| 853 |
+
"<extra_id_24>",
|
| 854 |
+
"<extra_id_25>",
|
| 855 |
+
"<extra_id_26>",
|
| 856 |
+
"<extra_id_27>",
|
| 857 |
+
"<extra_id_28>",
|
| 858 |
+
"<extra_id_29>",
|
| 859 |
+
"<extra_id_30>",
|
| 860 |
+
"<extra_id_31>",
|
| 861 |
+
"<extra_id_32>",
|
| 862 |
+
"<extra_id_33>",
|
| 863 |
+
"<extra_id_34>",
|
| 864 |
+
"<extra_id_35>",
|
| 865 |
+
"<extra_id_36>",
|
| 866 |
+
"<extra_id_37>",
|
| 867 |
+
"<extra_id_38>",
|
| 868 |
+
"<extra_id_39>",
|
| 869 |
+
"<extra_id_40>",
|
| 870 |
+
"<extra_id_41>",
|
| 871 |
+
"<extra_id_42>",
|
| 872 |
+
"<extra_id_43>",
|
| 873 |
+
"<extra_id_44>",
|
| 874 |
+
"<extra_id_45>",
|
| 875 |
+
"<extra_id_46>",
|
| 876 |
+
"<extra_id_47>",
|
| 877 |
+
"<extra_id_48>",
|
| 878 |
+
"<extra_id_49>",
|
| 879 |
+
"<extra_id_50>",
|
| 880 |
+
"<extra_id_51>",
|
| 881 |
+
"<extra_id_52>",
|
| 882 |
+
"<extra_id_53>",
|
| 883 |
+
"<extra_id_54>",
|
| 884 |
+
"<extra_id_55>",
|
| 885 |
+
"<extra_id_56>",
|
| 886 |
+
"<extra_id_57>",
|
| 887 |
+
"<extra_id_58>",
|
| 888 |
+
"<extra_id_59>",
|
| 889 |
+
"<extra_id_60>",
|
| 890 |
+
"<extra_id_61>",
|
| 891 |
+
"<extra_id_62>",
|
| 892 |
+
"<extra_id_63>",
|
| 893 |
+
"<extra_id_64>",
|
| 894 |
+
"<extra_id_65>",
|
| 895 |
+
"<extra_id_66>",
|
| 896 |
+
"<extra_id_67>",
|
| 897 |
+
"<extra_id_68>",
|
| 898 |
+
"<extra_id_69>",
|
| 899 |
+
"<extra_id_70>",
|
| 900 |
+
"<extra_id_71>",
|
| 901 |
+
"<extra_id_72>",
|
| 902 |
+
"<extra_id_73>",
|
| 903 |
+
"<extra_id_74>",
|
| 904 |
+
"<extra_id_75>",
|
| 905 |
+
"<extra_id_76>",
|
| 906 |
+
"<extra_id_77>",
|
| 907 |
+
"<extra_id_78>",
|
| 908 |
+
"<extra_id_79>",
|
| 909 |
+
"<extra_id_80>",
|
| 910 |
+
"<extra_id_81>",
|
| 911 |
+
"<extra_id_82>",
|
| 912 |
+
"<extra_id_83>",
|
| 913 |
+
"<extra_id_84>",
|
| 914 |
+
"<extra_id_85>",
|
| 915 |
+
"<extra_id_86>",
|
| 916 |
+
"<extra_id_87>",
|
| 917 |
+
"<extra_id_88>",
|
| 918 |
+
"<extra_id_89>",
|
| 919 |
+
"<extra_id_90>",
|
| 920 |
+
"<extra_id_91>",
|
| 921 |
+
"<extra_id_92>",
|
| 922 |
+
"<extra_id_93>",
|
| 923 |
+
"<extra_id_94>",
|
| 924 |
+
"<extra_id_95>",
|
| 925 |
+
"<extra_id_96>",
|
| 926 |
+
"<extra_id_97>",
|
| 927 |
+
"<extra_id_98>",
|
| 928 |
+
"<extra_id_99>"
|
| 929 |
+
],
|
| 930 |
+
"clean_up_tokenization_spaces": true,
|
| 931 |
+
"eos_token": "</s>",
|
| 932 |
+
"extra_ids": 100,
|
| 933 |
+
"model_max_length": 1000000000000000019884624838656,
|
| 934 |
+
"pad_token": "<pad>",
|
| 935 |
+
"sp_model_kwargs": {},
|
| 936 |
+
"tokenizer_class": "T5Tokenizer",
|
| 937 |
+
"unk_token": "<unk>"
|
| 938 |
+
}
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/refs/main
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
3fdb93344da77fbd75821e2fdd6307df3a0a1c96
|
demo/runorm_cache/models--RUNorm--RUNorm-normalizer-medium/refs/refs/pr/1
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
aaca4bd9f7f9e17390ca3cce913a8d14e61ed10c
|