Buckets:
| { | |
| "created_at": "2026-10-01T15:30:46.614515+00:00", | |
| "bucket_id": "HuggingFaceBio/Carbon-A-training-data", | |
| "visibility": "private", | |
| "model_id": "HuggingFaceBio/GENERanno-eukaryote-1.2b-cds-annotator", | |
| "tokenizer_revision": "49f69ea2085657034de291e02cb7791693c6872b", | |
| "evaluation_revision": "617613c8daa62345d79d0208b76b96149f6604af", | |
| "train_files": 4561, | |
| "test_files": 26, | |
| "train_bytes": 10155092707328, | |
| "test_bytes": 22126579712, | |
| "context_tokens": 16384, | |
| "bases_per_token": 6, | |
| "bases_per_strand": 98304, | |
| "label_transform": "nonzero to 1; zero to 0; concatenate plus then minus; special-token positions masked to -100", | |
| "evaluation_scope": "Original 26-genome evaluation panel used during training; not a new untouched test split.", | |
| "changes_to_arrays": "None; server-side copies by content hash.", | |
| "schema_inspection": "Headers and first rows sampled from each of the five training folders and one evaluation file; not a full corpus content audit." | |
| } | |
Xet Storage Details
- Size:
- 985 Bytes
- Xet hash:
- f959882cb7577ba893e7da85c89c8f6a61b1e03d4adba56df7d14815eae0d0b0
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.