robgreenberg3 commited on Feb 18

Commit

c8bcdd4

verified ·

1 Parent(s): 907b1bf

Upload folder using huggingface_hub

Browse files

Files changed (45) hide show

.gitattributes +10 -11
1_Pooling/config.json +7 -0
Prithvi-EO-V2-300M-TL-Sen1Floods11.pt +3 -0
Prithvi_EO_V2_300M_BurnScars.pt +3 -0
README.md +173 -0
burn_scars_config.yaml +104 -0
config.json +24 -0
config.yaml +154 -0
config_sentence_transformers.json +7 -0
data_config.json +1452 -0
examples/India_900498_S2Hand.tif +3 -0
examples/Spain_7370579_S2Hand.tif +3 -0
examples/USA_430764_S2Hand.tif +3 -0
examples/subsetted_512x512_HLS.S30.T10SEH.2018190.v1.4_merged.tif +3 -0
examples/subsetted_512x512_HLS.S30.T10SFF.2018190.v1.4_merged.tif +3 -0
examples/subsetted_512x512_HLS.S30.T10SGF.2020217.v1.4_merged.tif +3 -0
inference.py +335 -0
model.safetensors +3 -0
modules.json +20 -0
onnx/model.onnx +3 -0
onnx/model_O1.onnx +3 -0
onnx/model_O2.onnx +3 -0
onnx/model_O3.onnx +3 -0
onnx/model_O4.onnx +3 -0
onnx/model_qint8_arm64.onnx +3 -0
onnx/model_qint8_avx512.onnx +3 -0
onnx/model_qint8_avx512_vnni.onnx +3 -0
onnx/model_quint8_avx2.onnx +3 -0
openvino/openvino_model.bin +3 -0
openvino/openvino_model.xml +0 -0
openvino/openvino_model_qint8_quantized.bin +3 -0
openvino/openvino_model_qint8_quantized.xml +0 -0
pytorch_model.bin +3 -0
requirements.txt +6 -0
rust_model.ot +3 -0
sentence_bert_config.json +4 -0
special_tokens_map.json +1 -0
splits/test.txt +120 -0
splits/train.txt +524 -0
splits/val.txt +160 -0
tf_model.h5 +3 -0
tokenizer.json +0 -0
tokenizer_config.json +1 -0
train_script.py +344 -0
vocab.txt +0 -0

.gitattributes CHANGED Viewed

@@ -1,35 +1,34 @@
 *.7z filter=lfs diff=lfs merge=lfs -text
 *.arrow filter=lfs diff=lfs merge=lfs -text
 *.bin filter=lfs diff=lfs merge=lfs -text
 *.bz2 filter=lfs diff=lfs merge=lfs -text
-*.ckpt filter=lfs diff=lfs merge=lfs -text
 *.ftz filter=lfs diff=lfs merge=lfs -text
 *.gz filter=lfs diff=lfs merge=lfs -text
 *.h5 filter=lfs diff=lfs merge=lfs -text
 *.joblib filter=lfs diff=lfs merge=lfs -text
 *.lfs.* filter=lfs diff=lfs merge=lfs -text
-*.mlmodel filter=lfs diff=lfs merge=lfs -text
 *.model filter=lfs diff=lfs merge=lfs -text
 *.msgpack filter=lfs diff=lfs merge=lfs -text
-*.npy filter=lfs diff=lfs merge=lfs -text
-*.npz filter=lfs diff=lfs merge=lfs -text
 *.onnx filter=lfs diff=lfs merge=lfs -text
 *.ot filter=lfs diff=lfs merge=lfs -text
 *.parquet filter=lfs diff=lfs merge=lfs -text
 *.pb filter=lfs diff=lfs merge=lfs -text
-*.pickle filter=lfs diff=lfs merge=lfs -text
-*.pkl filter=lfs diff=lfs merge=lfs -text
 *.pt filter=lfs diff=lfs merge=lfs -text
 *.pth filter=lfs diff=lfs merge=lfs -text
 *.rar filter=lfs diff=lfs merge=lfs -text
-*.safetensors filter=lfs diff=lfs merge=lfs -text
-saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
-*.tar filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
-*.wasm filter=lfs diff=lfs merge=lfs -text
 *.xz filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
-*.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.7z filter=lfs diff=lfs merge=lfs -text
 *.arrow filter=lfs diff=lfs merge=lfs -text
 *.bin filter=lfs diff=lfs merge=lfs -text
+*.bin.* filter=lfs diff=lfs merge=lfs -text
 *.bz2 filter=lfs diff=lfs merge=lfs -text
 *.ftz filter=lfs diff=lfs merge=lfs -text
 *.gz filter=lfs diff=lfs merge=lfs -text
 *.h5 filter=lfs diff=lfs merge=lfs -text
 *.joblib filter=lfs diff=lfs merge=lfs -text
 *.lfs.* filter=lfs diff=lfs merge=lfs -text
 *.model filter=lfs diff=lfs merge=lfs -text
 *.msgpack filter=lfs diff=lfs merge=lfs -text
 *.onnx filter=lfs diff=lfs merge=lfs -text
 *.ot filter=lfs diff=lfs merge=lfs -text
 *.parquet filter=lfs diff=lfs merge=lfs -text
 *.pb filter=lfs diff=lfs merge=lfs -text
 *.pt filter=lfs diff=lfs merge=lfs -text
 *.pth filter=lfs diff=lfs merge=lfs -text
 *.rar filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.tar.* filter=lfs diff=lfs merge=lfs -text
 *.tflite filter=lfs diff=lfs merge=lfs -text
 *.tgz filter=lfs diff=lfs merge=lfs -text
 *.xz filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
+*.zstandard filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+model.safetensors filter=lfs diff=lfs merge=lfs -text
+examples/India_900498_S2Hand.tif filter=lfs diff=lfs merge=lfs -text
+examples/Spain_7370579_S2Hand.tif filter=lfs diff=lfs merge=lfs -text
+examples/USA_430764_S2Hand.tif filter=lfs diff=lfs merge=lfs -text
+examples/subsetted_512x512_HLS.S30.T10SEH.2018190.v1.4_merged.tif filter=lfs diff=lfs merge=lfs -text
+examples/subsetted_512x512_HLS.S30.T10SFF.2018190.v1.4_merged.tif filter=lfs diff=lfs merge=lfs -text
+examples/subsetted_512x512_HLS.S30.T10SGF.2020217.v1.4_merged.tif filter=lfs diff=lfs merge=lfs -text

1_Pooling/config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "word_embedding_dimension": 384,
+  "pooling_mode_cls_token": false,
+  "pooling_mode_mean_tokens": true,
+  "pooling_mode_max_tokens": false,
+  "pooling_mode_mean_sqrt_len_tokens": false
+}

Prithvi-EO-V2-300M-TL-Sen1Floods11.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3675e9c2b52547de8ff8a19f4881c28573e6d4d2f0805d866f4fc48c1e517d60
+size 1276843350

Prithvi_EO_V2_300M_BurnScars.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0c5f9334be9a75c9006387ab8f3dc05a55ea7fb5ef7956717316be57c62954d3
+size 1297798380

README.md ADDED Viewed

	@@ -0,0 +1,173 @@

+---
+language: en
+license: apache-2.0
+library_name: sentence-transformers
+tags:
+- sentence-transformers
+- feature-extraction
+- sentence-similarity
+- transformers
+datasets:
+- s2orc
+- flax-sentence-embeddings/stackexchange_xml
+- ms_marco
+- gooaq
+- yahoo_answers_topics
+- code_search_net
+- search_qa
+- eli5
+- snli
+- multi_nli
+- wikihow
+- natural_questions
+- trivia_qa
+- embedding-data/sentence-compression
+- embedding-data/flickr30k-captions
+- embedding-data/altlex
+- embedding-data/simple-wiki
+- embedding-data/QQP
+- embedding-data/SPECTER
+- embedding-data/PAQ_pairs
+- embedding-data/WikiAnswers
+pipeline_tag: sentence-similarity
+---
+# all-MiniLM-L6-v2
+This is a [sentence-transformers](https://www.SBERT.net) model: It maps sentences & paragraphs to a 384 dimensional dense vector space and can be used for tasks like clustering or semantic search.
+## Usage (Sentence-Transformers)
+Using this model becomes easy when you have [sentence-transformers](https://www.SBERT.net) installed:
+```
+pip install -U sentence-transformers
+```
+Then you can use the model like this:
+```python
+from sentence_transformers import SentenceTransformer
+sentences = ["This is an example sentence", "Each sentence is converted"]
+model = SentenceTransformer('sentence-transformers/all-MiniLM-L6-v2')
+embeddings = model.encode(sentences)
+print(embeddings)
+```
+## Usage (HuggingFace Transformers)
+Without [sentence-transformers](https://www.SBERT.net), you can use the model like this: First, you pass your input through the transformer model, then you have to apply the right pooling-operation on-top of the contextualized word embeddings.
+```python
+from transformers import AutoTokenizer, AutoModel
+import torch
+import torch.nn.functional as F
+#Mean Pooling - Take attention mask into account for correct averaging
+def mean_pooling(model_output, attention_mask):
+    token_embeddings = model_output[0] #First element of model_output contains all token embeddings
+    input_mask_expanded = attention_mask.unsqueeze(-1).expand(token_embeddings.size()).float()
+    return torch.sum(token_embeddings * input_mask_expanded, 1) / torch.clamp(input_mask_expanded.sum(1), min=1e-9)
+# Sentences we want sentence embeddings for
+sentences = ['This is an example sentence', 'Each sentence is converted']
+# Load model from HuggingFace Hub
+tokenizer = AutoTokenizer.from_pretrained('sentence-transformers/all-MiniLM-L6-v2')
+model = AutoModel.from_pretrained('sentence-transformers/all-MiniLM-L6-v2')
+# Tokenize sentences
+encoded_input = tokenizer(sentences, padding=True, truncation=True, return_tensors='pt')
+# Compute token embeddings
+with torch.no_grad():
+    model_output = model(**encoded_input)
+# Perform pooling
+sentence_embeddings = mean_pooling(model_output, encoded_input['attention_mask'])
+# Normalize embeddings
+sentence_embeddings = F.normalize(sentence_embeddings, p=2, dim=1)
+print("Sentence embeddings:")
+print(sentence_embeddings)
+```
+------
+## Background
+The project aims to train sentence embedding models on very large sentence level datasets using a self-supervised
+contrastive learning objective. We used the pretrained [`nreimers/MiniLM-L6-H384-uncased`](https://huggingface.co/nreimers/MiniLM-L6-H384-uncased) model and fine-tuned in on a
+1B sentence pairs dataset. We use a contrastive learning objective: given a sentence from the pair, the model should predict which out of a set of randomly sampled other sentences, was actually paired with it in our dataset.
+We developed this model during the
+[Community week using JAX/Flax for NLP & CV](https://discuss.huggingface.co/t/open-to-the-community-community-week-using-jax-flax-for-nlp-cv/7104),
+organized by Hugging Face. We developed this model as part of the project:
+[Train the Best Sentence Embedding Model Ever with 1B Training Pairs](https://discuss.huggingface.co/t/train-the-best-sentence-embedding-model-ever-with-1b-training-pairs/7354). We benefited from efficient hardware infrastructure to run the project: 7 TPUs v3-8, as well as intervention from Googles Flax, JAX, and Cloud team member about efficient deep learning frameworks.
+## Intended uses
+Our model is intended to be used as a sentence and short paragraph encoder. Given an input text, it outputs a vector which captures
+the semantic information. The sentence vector may be used for information retrieval, clustering or sentence similarity tasks.
+By default, input text longer than 256 word pieces is truncated.
+## Training procedure
+### Pre-training
+We use the pretrained [`nreimers/MiniLM-L6-H384-uncased`](https://huggingface.co/nreimers/MiniLM-L6-H384-uncased) model. Please refer to the model card for more detailed information about the pre-training procedure.
+### Fine-tuning
+We fine-tune the model using a contrastive objective. Formally, we compute the cosine similarity from each possible sentence pairs from the batch.
+We then apply the cross entropy loss by comparing with true pairs.
+#### Hyper parameters
+We trained our model on a TPU v3-8. We train the model during 100k steps using a batch size of 1024 (128 per TPU core).
+We use a learning rate warm up of 500. The sequence length was limited to 128 tokens. We used the AdamW optimizer with
+a 2e-5 learning rate. The full training script is accessible in this current repository: `train_script.py`.
+#### Training data
+We use the concatenation from multiple datasets to fine-tune our model. The total number of sentence pairs is above 1 billion sentences.
+We sampled each dataset given a weighted probability which configuration is detailed in the `data_config.json` file.
+| Dataset                                                  | Paper                                    | Number of training tuples  |
+|--------------------------------------------------------|:----------------------------------------:|:--------------------------:|
+| [Reddit comments (2015-2018)](https://github.com/PolyAI-LDN/conversational-datasets/tree/master/reddit) | [paper](https://arxiv.org/abs/1904.06472) | 726,484,430 |
+| [S2ORC](https://github.com/allenai/s2orc) Citation pairs (Abstracts) | [paper](https://aclanthology.org/2020.acl-main.447/) | 116,288,806 |
+| [WikiAnswers](https://github.com/afader/oqa#wikianswers-corpus) Duplicate question pairs | [paper](https://doi.org/10.1145/2623330.2623677) | 77,427,422 |
+| [PAQ](https://github.com/facebookresearch/PAQ) (Question, Answer) pairs | [paper](https://arxiv.org/abs/2102.07033) | 64,371,441 |
+| [S2ORC](https://github.com/allenai/s2orc) Citation pairs (Titles) | [paper](https://aclanthology.org/2020.acl-main.447/) | 52,603,982 |
+| [S2ORC](https://github.com/allenai/s2orc) (Title, Abstract) | [paper](https://aclanthology.org/2020.acl-main.447/) | 41,769,185 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) (Title, Body) pairs  | - | 25,316,456 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) (Title+Body, Answer) pairs  | - | 21,396,559 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) (Title, Answer) pairs  | - | 21,396,559 |
+| [MS MARCO](https://microsoft.github.io/msmarco/) triplets | [paper](https://doi.org/10.1145/3404835.3462804) | 9,144,553 |
+| [GOOAQ: Open Question Answering with Diverse Answer Types](https://github.com/allenai/gooaq) | [paper](https://arxiv.org/pdf/2104.08727.pdf) | 3,012,496 |
+| [Yahoo Answers](https://www.kaggle.com/soumikrakshit/yahoo-answers-dataset) (Title, Answer) | [paper](https://proceedings.neurips.cc/paper/2015/hash/250cf8b51c773f3f8dc8b4be867a9a02-Abstract.html) | 1,198,260 |
+| [Code Search](https://huggingface.co/datasets/code_search_net) | - | 1,151,414 |
+| [COCO](https://cocodataset.org/#home) Image captions | [paper](https://link.springer.com/chapter/10.1007%2F978-3-319-10602-1_48) | 828,395|
+| [SPECTER](https://github.com/allenai/specter) citation triplets | [paper](https://doi.org/10.18653/v1/2020.acl-main.207) | 684,100 |
+| [Yahoo Answers](https://www.kaggle.com/soumikrakshit/yahoo-answers-dataset) (Question, Answer) | [paper](https://proceedings.neurips.cc/paper/2015/hash/250cf8b51c773f3f8dc8b4be867a9a02-Abstract.html) | 681,164 |
+| [Yahoo Answers](https://www.kaggle.com/soumikrakshit/yahoo-answers-dataset) (Title, Question) | [paper](https://proceedings.neurips.cc/paper/2015/hash/250cf8b51c773f3f8dc8b4be867a9a02-Abstract.html) | 659,896 |
+| [SearchQA](https://huggingface.co/datasets/search_qa) | [paper](https://arxiv.org/abs/1704.05179) | 582,261 |
+| [Eli5](https://huggingface.co/datasets/eli5) | [paper](https://doi.org/10.18653/v1/p19-1346) | 325,475 |
+| [Flickr 30k](https://shannon.cs.illinois.edu/DenotationGraph/) | [paper](https://transacl.org/ojs/index.php/tacl/article/view/229/33) | 317,695 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) Duplicate questions (titles) | | 304,525 |
+| AllNLI ([SNLI](https://nlp.stanford.edu/projects/snli/) and [MultiNLI](https://cims.nyu.edu/~sbowman/multinli/) | [paper SNLI](https://doi.org/10.18653/v1/d15-1075), [paper MultiNLI](https://doi.org/10.18653/v1/n18-1101) | 277,230 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) Duplicate questions (bodies) | | 250,519 |
+| [Stack Exchange](https://huggingface.co/datasets/flax-sentence-embeddings/stackexchange_xml) Duplicate questions (titles+bodies) | | 250,460 |
+| [Sentence Compression](https://github.com/google-research-datasets/sentence-compression) | [paper](https://www.aclweb.org/anthology/D13-1155/) | 180,000 |
+| [Wikihow](https://github.com/pvl/wikihow_pairs_dataset) | [paper](https://arxiv.org/abs/1810.09305) | 128,542 |
+| [Altlex](https://github.com/chridey/altlex/) | [paper](https://aclanthology.org/P16-1135.pdf) | 112,696 |
+| [Quora Question Triplets](https://quoradata.quora.com/First-Quora-Dataset-Release-Question-Pairs) | - | 103,663 |
+| [Simple Wikipedia](https://cs.pomona.edu/~dkauchak/simplification/) | [paper](https://www.aclweb.org/anthology/P11-2117/) | 102,225 |
+| [Natural Questions (NQ)](https://ai.google.com/research/NaturalQuestions) | [paper](https://transacl.org/ojs/index.php/tacl/article/view/1455) | 100,231 |
+| [SQuAD2.0](https://rajpurkar.github.io/SQuAD-explorer/) | [paper](https://aclanthology.org/P18-2124.pdf) | 87,599 |
+| [TriviaQA](https://huggingface.co/datasets/trivia_qa) | - | 73,346 |
+| **Total** | | **1,170,060,424** |

burn_scars_config.yaml ADDED Viewed

	@@ -0,0 +1,104 @@

+# lightning.pytorch==2.4.0
+seed_everything: 2
+trainer:
+  logger: true
+  max_epochs: 100
+  log_every_n_steps: 1
+  callbacks:
+    - class_path: EarlyStopping
+      init_args:
+        monitor: val/loss
+        patience: 15
+    - class_path: LearningRateMonitor
+      init_args:
+        logging_interval: epoch
+  enable_progress_bar: false
+  precision: bf16-mixed
+model:
+  class_path: terratorch.tasks.SemanticSegmentationTask
+  init_args:
+    model_factory: EncoderDecoderFactory
+    model_args:
+      backbone: prithvi_eo_v2_300
+      backbone_pretrained: true
+      backbone_bands: ["BLUE", "GREEN", "RED", "NIR_NARROW", "SWIR_1", "SWIR_2"]
+      necks:
+        - name: SelectIndices
+          indices: [5, 11, 17, 23]
+        - name: ReshapeTokensToImage
+        - name: LearnedInterpolateToPyramidal
+      decoder: UNetDecoder
+      decoder_channels: [512, 256, 128, 64]
+      num_classes: 2
+    loss: ce
+    ignore_index: -1
+    freeze_backbone: false
+    plot_on_val: false
+    class_names: [Not burned, Burn scar]
+optimizer:
+  class_path: torch.optim.AdamW
+  init_args:
+    lr: 1.e-4
+lr_scheduler:
+  class_path: ReduceLROnPlateau
+  init_args:
+    monitor: val/loss
+    factor: 0.5
+    patience: 4
+data:
+  class_path: GenericNonGeoSegmentationDataModule
+  init_args:
+    batch_size: 8
+    num_workers: 8
+    dataset_bands:  # Dataset bands
+      - BLUE
+      - GREEN
+      - RED
+      - NIR_NARROW
+      - SWIR_1
+      - SWIR_2
+    output_bands: # Model input bands
+      - BLUE
+      - GREEN
+      - RED
+      - NIR_NARROW
+      - SWIR_1
+      - SWIR_2
+    rgb_indices:
+      - 2
+      - 1
+      - 0
+    train_data_root: hls_burn_scars/data
+    val_data_root: hls_burn_scars/data
+    test_data_root: hls_burn_scars/data
+    train_split: hls_burn_scars/splits/train.txt
+    val_split: hls_burn_scars/splits/val.txt
+    test_split: hls_burn_scars/splits/test.txt
+    img_grep: "*_merged.tif"
+    label_grep: "*.mask.tif"
+    means:
+      -  0.033349706741586264
+      -  0.05701185520536176
+      -  0.05889748132001316
+      -  0.2323245113436119
+      -  0.1972854853760658
+      -  0.11944914225186566
+    stds:
+      -  0.02269135568823774
+      -  0.026807560223070237
+      -  0.04004109844362779
+      -  0.07791732423672691
+      -  0.08708738838140137
+      -  0.07241979477437814
+    num_classes: 2
+    train_transform:
+      - class_path: albumentations.D4
+      - class_path: ToTensorV2
+    test_transform:
+      - class_path: ToTensorV2
+    no_data_replace: 0
+    no_label_replace: -1

config.json ADDED Viewed

	@@ -0,0 +1,24 @@

+{
+  "_name_or_path": "nreimers/MiniLM-L6-H384-uncased",
+  "architectures": [
+    "BertModel"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "gradient_checkpointing": false,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 384,
+  "initializer_range": 0.02,
+  "intermediate_size": 1536,
+  "layer_norm_eps": 1e-12,
+  "max_position_embeddings": 512,
+  "model_type": "bert",
+  "num_attention_heads": 12,
+  "num_hidden_layers": 6,
+  "pad_token_id": 0,
+  "position_embedding_type": "absolute",
+  "transformers_version": "4.8.2",
+  "type_vocab_size": 2,
+  "use_cache": true,
+  "vocab_size": 30522
+}

config.yaml ADDED Viewed

	@@ -0,0 +1,154 @@

+# lightning.pytorch==2.4.0
+seed_everything: 0
+trainer:
+  accelerator: auto
+  strategy: auto
+  devices: auto
+  num_nodes: 1
+  precision: 16-mixed
+  logger: true
+  callbacks:
+  - class_path: lightning.pytorch.callbacks.RichProgressBar
+    init_args:
+      refresh_rate: 1
+      leave: false
+      theme:
+        description: white
+        progress_bar: '#6206E0'
+        progress_bar_finished: '#6206E0'
+        progress_bar_pulse: '#6206E0'
+        batch_progress: white
+        time: grey54
+        processing_speed: grey70
+        metrics: white
+        metrics_text_delimiter: ' '
+        metrics_format: .3f
+  - class_path: lightning.pytorch.callbacks.LearningRateMonitor
+    init_args:
+      logging_interval: epoch
+      log_momentum: false
+      log_weight_decay: false
+  - class_path: lightning.pytorch.callbacks.EarlyStopping
+    init_args:
+      monitor: val/loss
+      min_delta: 0.0
+      patience: 20
+      verbose: false
+      mode: min
+      strict: true
+      check_finite: true
+      log_rank_zero_only: false
+  fast_dev_run: false
+  max_epochs: 50
+  max_steps: -1
+  overfit_batches: 0.0
+  check_val_every_n_epoch: 2
+  log_every_n_steps: 10
+  enable_checkpointing: true
+  accumulate_grad_batches: 1
+  inference_mode: true
+  use_distributed_sampler: true
+  detect_anomaly: false
+  barebones: false
+  sync_batchnorm: false
+  reload_dataloaders_every_n_epochs: 0
+  default_root_dir: /dccstor/geofm-finetuning/benchmark-geo-bench-paolo/
+model:
+  class_path: terratorch.tasks.SemanticSegmentationTask
+  init_args:
+    model_args:
+      backbone_pretrained: true
+      backbone: prithvi_eo_v2_300_tl
+      decoder: UperNetDecoder
+      decoder_channels: 256
+      decoder_scale_modules: true
+      num_classes: 2
+      rescale: true
+      backbone_bands:
+      - BLUE
+      - GREEN
+      - RED
+      - NIR_NARROW
+      - SWIR_1
+      - SWIR_2
+      head_dropout: 0.1
+      necks:
+      - name: SelectIndices
+        indices:
+        - 5
+        - 11
+        - 17
+        - 23
+      - name: ReshapeTokensToImage
+    model_factory: EncoderDecoderFactory
+    loss: ce
+    ignore_index: -1
+    lr: 0.001
+    freeze_backbone: false
+    freeze_decoder: false
+    plot_on_val: 10
+data:
+  class_path: terratorch.datamodules.Sen1Floods11NonGeoDataModule
+  init_args:
+    data_root: /dccstor/geofm-finetuning/datasets/sen1floods11
+    batch_size: 16
+    num_workers: 8
+    bands:
+    - BLUE
+    - GREEN
+    - RED
+    - NIR_NARROW
+    - SWIR_1
+    - SWIR_2
+    train_transform:
+    - class_path: albumentations.RandomCrop
+      init_args:
+        height: 224
+        width: 224
+        p: 1.0
+    - class_path: albumentations.HorizontalFlip
+      init_args:
+        p: 0.5
+    - class_path: albumentations.VerticalFlip
+      init_args:
+        p: 0.5
+    - class_path: albumentations.pytorch.ToTensorV2
+      init_args:
+        transpose_mask: false
+        p: 1.0
+    val_transform:
+    - class_path: albumentations.pytorch.ToTensorV2
+      init_args:
+        transpose_mask: false
+        p: 1.0
+    test_transform:
+    - class_path: albumentations.pytorch.ToTensorV2
+      init_args:
+        transpose_mask: false
+        p: 1.0
+    drop_last: true
+    constant_scale: 0.0001
+    no_data_replace: 0.0
+    no_label_replace: -1
+    use_metadata: false
+out_dtype: int16
+deploy_config_file: true
+optimizer:
+  class_path: torch.optim.AdamW
+  init_args:
+    lr: 5.0e-05
+    betas:
+    - 0.9
+    - 0.999
+    eps: 1.0e-08
+    weight_decay: 0.05
+    amsgrad: false
+    maximize: false
+    capturable: false
+    differentiable: false
+lr_scheduler:
+  class_path: torch.optim.lr_scheduler.CosineAnnealingLR
+  init_args:
+    T_max: 50
+    eta_min: 0
+    last_epoch: -1

config_sentence_transformers.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "__version__": {
+    "sentence_transformers": "2.0.0",
+    "transformers": "4.6.1",
+    "pytorch": "1.8.1"
+  }
+}

data_config.json ADDED Viewed

	@@ -0,0 +1,1452 @@

+[
+    {
+        "name": "stackexchange_title_body/skeptics.stackexchange.com.jsonl.gz",
+        "lines": 10009,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/islam.stackexchange.com.jsonl.gz",
+        "lines": 10052,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/islam.stackexchange.com.jsonl.gz",
+        "lines": 10052,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/anime.stackexchange.com.jsonl.gz",
+        "lines": 10131,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/anime.stackexchange.com.jsonl.gz",
+        "lines": 10131,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/writers.stackexchange.com.jsonl.gz",
+        "lines": 10157,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/astronomy.stackexchange.com.jsonl.gz",
+        "lines": 10462,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/vi.stackexchange.com.jsonl.gz",
+        "lines": 10551,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/french.stackexchange.com.jsonl.gz",
+        "lines": 10578,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/french.stackexchange.com.jsonl.gz",
+        "lines": 10578,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/cstheory.stackexchange.com.jsonl.gz",
+        "lines": 10642,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/civicrm.stackexchange.com.jsonl.gz",
+        "lines": 10648,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/civicrm.stackexchange.com.jsonl.gz",
+        "lines": 10648,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/expressionengine.stackexchange.com.jsonl.gz",
+        "lines": 10742,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/expressionengine.stackexchange.com.jsonl.gz",
+        "lines": 10742,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/engineering.stackexchange.com.jsonl.gz",
+        "lines": 10753,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/history.stackexchange.com.jsonl.gz",
+        "lines": 10766,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/history.stackexchange.com.jsonl.gz",
+        "lines": 10766,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/french.stackexchange.com.jsonl.gz",
+        "lines": 10794,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/politics.stackexchange.com.jsonl.gz",
+        "lines": 11047,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/politics.stackexchange.com.jsonl.gz",
+        "lines": 11047,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/economics.stackexchange.com.jsonl.gz",
+        "lines": 11115,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/craftcms.stackexchange.com.jsonl.gz",
+        "lines": 11236,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/craftcms.stackexchange.com.jsonl.gz",
+        "lines": 11236,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/anime.stackexchange.com.jsonl.gz",
+        "lines": 11444,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/christianity.stackexchange.com.jsonl.gz",
+        "lines": 11498,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/christianity.stackexchange.com.jsonl.gz",
+        "lines": 11498,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/softwarerecs.stackexchange.com.jsonl.gz",
+        "lines": 11761,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/softwarerecs.stackexchange.com.jsonl.gz",
+        "lines": 11761,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/boardgames.stackexchange.com.jsonl.gz",
+        "lines": 11805,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_Title_Answer/boardgames.stackexchange.com.jsonl.gz",
+        "lines": 11805,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/islam.stackexchange.com.jsonl.gz",
+        "lines": 11853,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/expressionengine.stackexchange.com.jsonl.gz",
+        "lines": 11866,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/politics.stackexchange.com.jsonl.gz",
+        "lines": 11894,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/history.stackexchange.com.jsonl.gz",
+        "lines": 12021,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/christianity.stackexchange.com.jsonl.gz",
+        "lines": 12108,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/boardgames.stackexchange.com.jsonl.gz",
+        "lines": 12149,
+        "weight": 1
+    },
+    {
+        "name": "flickr30k_captions.jsonl.gz",
+        "lines": 317695,
+        "weight": 1
+    },
+    {
+        "name": "coco_captions.jsonl.gz",
+        "lines": 828395,
+        "weight": 1
+    },
+    {
+        "name": "codesearchnet.jsonl.gz",
+        "lines": 1151414,
+        "weight": 1
+    },
+    {
+        "name": "stackexchange_title_body/civicrm.stackexchange.com.jsonl.gz",
+        "lines": 12543,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/craftcms.stackexchange.com.jsonl.gz",
+        "lines": 12574,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/networkengineering.stackexchange.com.jsonl.gz",
+        "lines": 12590,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/networkengineering.stackexchange.com.jsonl.gz",
+        "lines": 12590,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/space.stackexchange.com.jsonl.gz",
+        "lines": 12893,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/space.stackexchange.com.jsonl.gz",
+        "lines": 12893,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/quant.stackexchange.com.jsonl.gz",
+        "lines": 12933,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/quant.stackexchange.com.jsonl.gz",
+        "lines": 12933,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/philosophy.stackexchange.com.jsonl.gz",
+        "lines": 13114,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/philosophy.stackexchange.com.jsonl.gz",
+        "lines": 13114,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/gardening.stackexchange.com.jsonl.gz",
+        "lines": 13246,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/gardening.stackexchange.com.jsonl.gz",
+        "lines": 13246,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/hinduism.stackexchange.com.jsonl.gz",
+        "lines": 13450,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/networkengineering.stackexchange.com.jsonl.gz",
+        "lines": 13454,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/german.stackexchange.com.jsonl.gz",
+        "lines": 13733,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/german.stackexchange.com.jsonl.gz",
+        "lines": 13733,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/german.stackexchange.com.jsonl.gz",
+        "lines": 13950,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/philosophy.stackexchange.com.jsonl.gz",
+        "lines": 14829,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/gardening.stackexchange.com.jsonl.gz",
+        "lines": 15136,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/space.stackexchange.com.jsonl.gz",
+        "lines": 15142,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/bicycles.stackexchange.com.jsonl.gz",
+        "lines": 15708,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/bicycles.stackexchange.com.jsonl.gz",
+        "lines": 15708,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/law.stackexchange.com.jsonl.gz",
+        "lines": 16133,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/law.stackexchange.com.jsonl.gz",
+        "lines": 16133,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/arduino.stackexchange.com.jsonl.gz",
+        "lines": 16281,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/arduino.stackexchange.com.jsonl.gz",
+        "lines": 16281,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/bicycles.stackexchange.com.jsonl.gz",
+        "lines": 16353,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/emacs.stackexchange.com.jsonl.gz",
+        "lines": 16830,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/emacs.stackexchange.com.jsonl.gz",
+        "lines": 16830,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/quant.stackexchange.com.jsonl.gz",
+        "lines": 17261,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/dsp.stackexchange.com.jsonl.gz",
+        "lines": 17430,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/dsp.stackexchange.com.jsonl.gz",
+        "lines": 17430,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/puzzling.stackexchange.com.jsonl.gz",
+        "lines": 17448,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/puzzling.stackexchange.com.jsonl.gz",
+        "lines": 17448,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/puzzling.stackexchange.com.jsonl.gz",
+        "lines": 17851,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/law.stackexchange.com.jsonl.gz",
+        "lines": 17941,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/movies.stackexchange.com.jsonl.gz",
+        "lines": 18243,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/movies.stackexchange.com.jsonl.gz",
+        "lines": 18243,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/mechanics.stackexchange.com.jsonl.gz",
+        "lines": 18613,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/mechanics.stackexchange.com.jsonl.gz",
+        "lines": 18613,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/aviation.stackexchange.com.jsonl.gz",
+        "lines": 18755,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/aviation.stackexchange.com.jsonl.gz",
+        "lines": 18755,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/biology.stackexchange.com.jsonl.gz",
+        "lines": 19277,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/biology.stackexchange.com.jsonl.gz",
+        "lines": 19277,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/crypto.stackexchange.com.jsonl.gz",
+        "lines": 19404,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/crypto.stackexchange.com.jsonl.gz",
+        "lines": 19404,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/arduino.stackexchange.com.jsonl.gz",
+        "lines": 19553,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/music.stackexchange.com.jsonl.gz",
+        "lines": 19936,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/music.stackexchange.com.jsonl.gz",
+        "lines": 19936,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/aviation.stackexchange.com.jsonl.gz",
+        "lines": 20139,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/softwarerecs.stackexchange.com.jsonl.gz",
+        "lines": 20142,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/movies.stackexchange.com.jsonl.gz",
+        "lines": 20181,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/datascience.stackexchange.com.jsonl.gz",
+        "lines": 20503,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/datascience.stackexchange.com.jsonl.gz",
+        "lines": 20503,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/music.stackexchange.com.jsonl.gz",
+        "lines": 20636,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/japanese.stackexchange.com.jsonl.gz",
+        "lines": 20948,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/japanese.stackexchange.com.jsonl.gz",
+        "lines": 20948,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/emacs.stackexchange.com.jsonl.gz",
+        "lines": 21055,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/dsp.stackexchange.com.jsonl.gz",
+        "lines": 21252,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/japanese.stackexchange.com.jsonl.gz",
+        "lines": 22056,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/bitcoin.stackexchange.com.jsonl.gz",
+        "lines": 22474,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/bitcoin.stackexchange.com.jsonl.gz",
+        "lines": 22474,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/cooking.stackexchange.com.jsonl.gz",
+        "lines": 22641,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/cooking.stackexchange.com.jsonl.gz",
+        "lines": 22641,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/mechanics.stackexchange.com.jsonl.gz",
+        "lines": 22868,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/photo.stackexchange.com.jsonl.gz",
+        "lines": 23204,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/photo.stackexchange.com.jsonl.gz",
+        "lines": 23204,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/crypto.stackexchange.com.jsonl.gz",
+        "lines": 23231,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/cooking.stackexchange.com.jsonl.gz",
+        "lines": 23705,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/photo.stackexchange.com.jsonl.gz",
+        "lines": 23753,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/workplace.stackexchange.com.jsonl.gz",
+        "lines": 24012,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/workplace.stackexchange.com.jsonl.gz",
+        "lines": 24012,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/meta.stackoverflow.com.jsonl.gz",
+        "lines": 24044,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/meta.stackoverflow.com.jsonl.gz",
+        "lines": 24044,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/raspberrypi.stackexchange.com.jsonl.gz",
+        "lines": 24143,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_Title_Answer/raspberrypi.stackexchange.com.jsonl.gz",
+        "lines": 24143,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/workplace.stackexchange.com.jsonl.gz",
+        "lines": 24189,
+        "weight": 2
+    },
+    {
+        "name": "stackexchange_title_body/biology.stackexchange.com.jsonl.gz",
+        "lines": 24447,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/webapps.stackexchange.com.jsonl.gz",
+        "lines": 24867,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/webapps.stackexchange.com.jsonl.gz",
+        "lines": 24867,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/bitcoin.stackexchange.com.jsonl.gz",
+        "lines": 25374,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/judaism.stackexchange.com.jsonl.gz",
+        "lines": 26085,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/judaism.stackexchange.com.jsonl.gz",
+        "lines": 26085,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/ethereum.stackexchange.com.jsonl.gz",
+        "lines": 26124,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/ethereum.stackexchange.com.jsonl.gz",
+        "lines": 26124,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/worldbuilding.stackexchange.com.jsonl.gz",
+        "lines": 26210,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/worldbuilding.stackexchange.com.jsonl.gz",
+        "lines": 26210,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/worldbuilding.stackexchange.com.jsonl.gz",
+        "lines": 26763,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/chemistry.stackexchange.com.jsonl.gz",
+        "lines": 27061,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/chemistry.stackexchange.com.jsonl.gz",
+        "lines": 27061,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/datascience.stackexchange.com.jsonl.gz",
+        "lines": 27397,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/graphicdesign.stackexchange.com.jsonl.gz",
+        "lines": 28083,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/graphicdesign.stackexchange.com.jsonl.gz",
+        "lines": 28083,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/ux.stackexchange.com.jsonl.gz",
+        "lines": 28901,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/ux.stackexchange.com.jsonl.gz",
+        "lines": 28901,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/ux.stackexchange.com.jsonl.gz",
+        "lines": 29403,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/money.stackexchange.com.jsonl.gz",
+        "lines": 29404,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/money.stackexchange.com.jsonl.gz",
+        "lines": 29404,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/webapps.stackexchange.com.jsonl.gz",
+        "lines": 29697,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/cs.stackexchange.com.jsonl.gz",
+        "lines": 30010,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/cs.stackexchange.com.jsonl.gz",
+        "lines": 30010,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/graphicdesign.stackexchange.com.jsonl.gz",
+        "lines": 30233,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/webmasters.stackexchange.com.jsonl.gz",
+        "lines": 30370,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/webmasters.stackexchange.com.jsonl.gz",
+        "lines": 30370,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/raspberrypi.stackexchange.com.jsonl.gz",
+        "lines": 30625,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/money.stackexchange.com.jsonl.gz",
+        "lines": 32021,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/judaism.stackexchange.com.jsonl.gz",
+        "lines": 32028,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/academia.stackexchange.com.jsonl.gz",
+        "lines": 32137,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_Title_Answer/academia.stackexchange.com.jsonl.gz",
+        "lines": 32137,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/ethereum.stackexchange.com.jsonl.gz",
+        "lines": 32760,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/academia.stackexchange.com.jsonl.gz",
+        "lines": 34331,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/chemistry.stackexchange.com.jsonl.gz",
+        "lines": 34506,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/webmasters.stackexchange.com.jsonl.gz",
+        "lines": 34559,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_title_body/meta.stackoverflow.com.jsonl.gz",
+        "lines": 36456,
+        "weight": 3
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/travel.stackexchange.com.jsonl.gz",
+        "lines": 36533,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_Title_Answer/travel.stackexchange.com.jsonl.gz",
+        "lines": 36533,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/android.stackexchange.com.jsonl.gz",
+        "lines": 38077,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_Title_Answer/android.stackexchange.com.jsonl.gz",
+        "lines": 38077,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_title_body/cs.stackexchange.com.jsonl.gz",
+        "lines": 38314,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/gamedev.stackexchange.com.jsonl.gz",
+        "lines": 40154,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_Title_Answer/gamedev.stackexchange.com.jsonl.gz",
+        "lines": 40154,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/rpg.stackexchange.com.jsonl.gz",
+        "lines": 40435,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_Title_Answer/rpg.stackexchange.com.jsonl.gz",
+        "lines": 40435,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_title_body/travel.stackexchange.com.jsonl.gz",
+        "lines": 41227,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/codereview.stackexchange.com.jsonl.gz",
+        "lines": 41748,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_Title_Answer/codereview.stackexchange.com.jsonl.gz",
+        "lines": 41748,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_title_body/rpg.stackexchange.com.jsonl.gz",
+        "lines": 42303,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_title_body/codereview.stackexchange.com.jsonl.gz",
+        "lines": 45765,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_title_body/gamedev.stackexchange.com.jsonl.gz",
+        "lines": 46485,
+        "weight": 4
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/softwareengineering.stackexchange.com.jsonl.gz",
+        "lines": 51326,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/softwareengineering.stackexchange.com.jsonl.gz",
+        "lines": 51326,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/security.stackexchange.com.jsonl.gz",
+        "lines": 51355,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/security.stackexchange.com.jsonl.gz",
+        "lines": 51355,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_title_body/android.stackexchange.com.jsonl.gz",
+        "lines": 51608,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/diy.stackexchange.com.jsonl.gz",
+        "lines": 52896,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/diy.stackexchange.com.jsonl.gz",
+        "lines": 52896,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_title_body/softwareengineering.stackexchange.com.jsonl.gz",
+        "lines": 53942,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/blender.stackexchange.com.jsonl.gz",
+        "lines": 54153,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/blender.stackexchange.com.jsonl.gz",
+        "lines": 54153,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/scifi.stackexchange.com.jsonl.gz",
+        "lines": 54805,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/scifi.stackexchange.com.jsonl.gz",
+        "lines": 54805,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_title_body/security.stackexchange.com.jsonl.gz",
+        "lines": 58000,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/mathematica.stackexchange.com.jsonl.gz",
+        "lines": 59895,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/mathematica.stackexchange.com.jsonl.gz",
+        "lines": 59895,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_title_body/diy.stackexchange.com.jsonl.gz",
+        "lines": 60083,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/meta.stackexchange.com.jsonl.gz",
+        "lines": 60744,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_Title_Answer/meta.stackexchange.com.jsonl.gz",
+        "lines": 60744,
+        "weight": 5
+    },
+    {
+        "name": "stackexchange_title_body/scifi.stackexchange.com.jsonl.gz",
+        "lines": 61528,
+        "weight": 6
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/drupal.stackexchange.com.jsonl.gz",
+        "lines": 67817,
+        "weight": 6
+    },
+    {
+        "name": "stackexchange_Title_Answer/drupal.stackexchange.com.jsonl.gz",
+        "lines": 67817,
+        "weight": 6
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/dba.stackexchange.com.jsonl.gz",
+        "lines": 71449,
+        "weight": 6
+    },
+    {
+        "name": "stackexchange_Title_Answer/dba.stackexchange.com.jsonl.gz",
+        "lines": 71449,
+        "weight": 6
+    },
+    {
+        "name": "stackexchange_title_body/mathematica.stackexchange.com.jsonl.gz",
+        "lines": 73131,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/ell.stackexchange.com.jsonl.gz",
+        "lines": 77892,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_Title_Answer/ell.stackexchange.com.jsonl.gz",
+        "lines": 77892,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/magento.stackexchange.com.jsonl.gz",
+        "lines": 79241,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_Title_Answer/magento.stackexchange.com.jsonl.gz",
+        "lines": 79241,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_title_body/drupal.stackexchange.com.jsonl.gz",
+        "lines": 79717,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/sharepoint.stackexchange.com.jsonl.gz",
+        "lines": 80420,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_Title_Answer/sharepoint.stackexchange.com.jsonl.gz",
+        "lines": 80420,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_title_body/blender.stackexchange.com.jsonl.gz",
+        "lines": 80766,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_title_body/dba.stackexchange.com.jsonl.gz",
+        "lines": 81871,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/gaming.stackexchange.com.jsonl.gz",
+        "lines": 82887,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_Title_Answer/gaming.stackexchange.com.jsonl.gz",
+        "lines": 82887,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_title_body/ell.stackexchange.com.jsonl.gz",
+        "lines": 83271,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_title_body/meta.stackexchange.com.jsonl.gz",
+        "lines": 83510,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/wordpress.stackexchange.com.jsonl.gz",
+        "lines": 83621,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_Title_Answer/wordpress.stackexchange.com.jsonl.gz",
+        "lines": 83621,
+        "weight": 7
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/mathoverflow.net.jsonl.gz",
+        "lines": 85289,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_Title_Answer/mathoverflow.net.jsonl.gz",
+        "lines": 85289,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/salesforce.stackexchange.com.jsonl.gz",
+        "lines": 87272,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_Title_Answer/salesforce.stackexchange.com.jsonl.gz",
+        "lines": 87272,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_title_body/gaming.stackexchange.com.jsonl.gz",
+        "lines": 88912,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/apple.stackexchange.com.jsonl.gz",
+        "lines": 92487,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_Title_Answer/apple.stackexchange.com.jsonl.gz",
+        "lines": 92487,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_title_body/sharepoint.stackexchange.com.jsonl.gz",
+        "lines": 94011,
+        "weight": 8
+    },
+    {
+        "name": "stackexchange_title_body/magento.stackexchange.com.jsonl.gz",
+        "lines": 99991,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/gis.stackexchange.com.jsonl.gz",
+        "lines": 100254,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_Title_Answer/gis.stackexchange.com.jsonl.gz",
+        "lines": 100254,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_title_body/wordpress.stackexchange.com.jsonl.gz",
+        "lines": 100474,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/english.stackexchange.com.jsonl.gz",
+        "lines": 100640,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_Title_Answer/english.stackexchange.com.jsonl.gz",
+        "lines": 100640,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_title_body/salesforce.stackexchange.com.jsonl.gz",
+        "lines": 105260,
+        "weight": 9
+    },
+    {
+        "name": "stackexchange_title_body/english.stackexchange.com.jsonl.gz",
+        "lines": 109522,
+        "weight": 10
+    },
+    {
+        "name": "stackexchange_title_body/apple.stackexchange.com.jsonl.gz",
+        "lines": 110622,
+        "weight": 10
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/stats.stackexchange.com.jsonl.gz",
+        "lines": 115679,
+        "weight": 10
+    },
+    {
+        "name": "stackexchange_Title_Answer/stats.stackexchange.com.jsonl.gz",
+        "lines": 115679,
+        "weight": 10
+    },
+    {
+        "name": "stackexchange_title_body/mathoverflow.net.jsonl.gz",
+        "lines": 120851,
+        "weight": 10
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/electronics.stackexchange.com.jsonl.gz",
+        "lines": 129494,
+        "weight": 11
+    },
+    {
+        "name": "stackexchange_Title_Answer/electronics.stackexchange.com.jsonl.gz",
+        "lines": 129494,
+        "weight": 11
+    },
+    {
+        "name": "stackexchange_title_body/gis.stackexchange.com.jsonl.gz",
+        "lines": 131000,
+        "weight": 11
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/physics.stackexchange.com.jsonl.gz",
+        "lines": 141230,
+        "weight": 12
+    },
+    {
+        "name": "stackexchange_Title_Answer/physics.stackexchange.com.jsonl.gz",
+        "lines": 141230,
+        "weight": 12
+    },
+    {
+        "name": "stackexchange_title_body/electronics.stackexchange.com.jsonl.gz",
+        "lines": 143582,
+        "weight": 12
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/unix.stackexchange.com.jsonl.gz",
+        "lines": 155414,
+        "weight": 13
+    },
+    {
+        "name": "stackexchange_Title_Answer/unix.stackexchange.com.jsonl.gz",
+        "lines": 155414,
+        "weight": 13
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/tex.stackexchange.com.jsonl.gz",
+        "lines": 171628,
+        "weight": 15
+    },
+    {
+        "name": "stackexchange_Title_Answer/tex.stackexchange.com.jsonl.gz",
+        "lines": 171628,
+        "weight": 15
+    },
+    {
+        "name": "stackexchange_title_body/physics.stackexchange.com.jsonl.gz",
+        "lines": 173307,
+        "weight": 15
+    },
+    {
+        "name": "stackexchange_title_body/stats.stackexchange.com.jsonl.gz",
+        "lines": 173466,
+        "weight": 15
+    },
+    {
+        "name": "stackexchange_title_body/unix.stackexchange.com.jsonl.gz",
+        "lines": 185997,
+        "weight": 16
+    },
+    {
+        "name": "stackexchange_title_body/tex.stackexchange.com.jsonl.gz",
+        "lines": 202954,
+        "weight": 17
+    },
+    {
+        "name": "TriviaQA_pairs.jsonl.gz",
+        "lines": 73346,
+        "weight": 19
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/serverfault.com.jsonl.gz",
+        "lines": 238507,
+        "weight": 20
+    },
+    {
+        "name": "stackexchange_Title_Answer/serverfault.com.jsonl.gz",
+        "lines": 238507,
+        "weight": 20
+    },
+    {
+        "name": "stackexchange_duplicate_questions_title-body_title-body.jsonl.gz",
+        "lines": 250460,
+        "weight": 21
+    },
+    {
+        "name": "stackexchange_duplicate_questions_body_body.jsonl.gz",
+        "lines": 250519,
+        "weight": 21
+    },
+    {
+        "name": "squad_pairs.jsonl.gz",
+        "lines": 87599,
+        "weight": 22
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/askubuntu.com.jsonl.gz",
+        "lines": 267135,
+        "weight": 22
+    },
+    {
+        "name": "stackexchange_Title_Answer/askubuntu.com.jsonl.gz",
+        "lines": 267135,
+        "weight": 22
+    },
+    {
+        "name": "stackexchange_title_body/serverfault.com.jsonl.gz",
+        "lines": 270904,
+        "weight": 23
+    },
+    {
+        "name": "NQ-train_pairs.jsonl.gz",
+        "lines": 100231,
+        "weight": 25
+    },
+    {
+        "name": "SimpleWiki.jsonl.gz",
+        "lines": 102225,
+        "weight": 26
+    },
+    {
+        "name": "quora_duplicates_triplets.jsonl.gz",
+        "lines": 103663,
+        "weight": 26
+    },
+    {
+        "name": "stackexchange_duplicate_questions_title_title.jsonl.gz",
+        "lines": 304525,
+        "weight": 26
+    },
+    {
+        "name": "altlex.jsonl.gz",
+        "lines": 112696,
+        "weight": 28
+    },
+    {
+        "name": "stackexchange_title_body/askubuntu.com.jsonl.gz",
+        "lines": 347925,
+        "weight": 29
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/superuser.com.jsonl.gz",
+        "lines": 352610,
+        "weight": 30
+    },
+    {
+        "name": "stackexchange_Title_Answer/superuser.com.jsonl.gz",
+        "lines": 352610,
+        "weight": 30
+    },
+    {
+        "name": "wikihow.jsonl.gz",
+        "lines": 128542,
+        "weight": 32
+    },
+    {
+        "name": "stackexchange_title_body/superuser.com.jsonl.gz",
+        "lines": 435463,
+        "weight": 36
+    },
+    {
+        "name": "stackexchange_title_body/small_stackexchanges.jsonl.gz",
+        "lines": 448146,
+        "weight": 37
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/small_stackexchanges.jsonl.gz",
+        "lines": 460256,
+        "weight": 38
+    },
+    {
+        "name": "stackexchange_Title_Answer/small_stackexchanges.jsonl.gz",
+        "lines": 460256,
+        "weight": 38
+    },
+    {
+        "name": "sentence-compression.jsonl.gz",
+        "lines": 180000,
+        "weight": 45
+    },
+    {
+        "name": "AllNLI.jsonl.gz",
+        "lines": 277230,
+        "weight": 69
+    },
+    {
+        "name": "eli5_question_answer.jsonl.gz",
+        "lines": 325475,
+        "weight": 81
+    },
+    {
+        "name": "reddit/reddit_2015.jsonl.gz",
+        "lines": 135108166,
+        "weight": 82
+    },
+    {
+        "name": "reddit/reddit_2016.jsonl.gz",
+        "lines": 159164386,
+        "weight": 82
+    },
+    {
+        "name": "reddit/reddit_2017.jsonl.gz",
+        "lines": 191485219,
+        "weight": 82
+    },
+    {
+        "name": "reddit/reddit_2018.jsonl.gz",
+        "lines": 240726659,
+        "weight": 82
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/math.stackexchange.com.jsonl.gz",
+        "lines": 1100953,
+        "weight": 83
+    },
+    {
+        "name": "stackexchange_Title_Answer/math.stackexchange.com.jsonl.gz",
+        "lines": 1100953,
+        "weight": 83
+    },
+    {
+        "name": "stackexchange_title_body/math.stackexchange.com.jsonl.gz",
+        "lines": 1338443,
+        "weight": 83
+    },
+    {
+        "name": "stackexchange_TitleBody_Answer/stackoverflow.com-Posts.jsonl.gz",
+        "lines": 15768211,
+        "weight": 83
+    },
+    {
+        "name": "stackexchange_Title_Answer/stackoverflow.com-Posts.jsonl.gz",
+        "lines": 15768211,
+        "weight": 83
+    },
+    {
+        "name": "stackexchange_title_body/stackoverflow.com-Posts.jsonl.gz",
+        "lines": 18562443,
+        "weight": 83
+    },
+    {
+        "name": "specter_train_triples.jsonl.gz",
+        "lines": 684100,
+        "weight": 84
+    },
+    {
+        "name": "S2ORC_title_abstract.jsonl.gz",
+        "lines": 41769185,
+        "weight": 123
+    },
+    {
+        "name": "S2ORC_citation_pairs.jsonl.gz",
+        "lines": 52603982,
+        "weight": 123
+    },
+    {
+        "name": "PAQ_pairs.jsonl.gz",
+        "lines": 64371441,
+        "weight": 123
+    },
+    {
+        "name": "WikiAnswers_pairs.jsonl.gz",
+        "lines": 77427422,
+        "weight": 123
+    },
+    {
+        "name": "S2ORC_citation_pairs_abstract.jsonl.gz",
+        "lines": 116288806,
+        "weight": 123
+    },
+    {
+        "name": "searchQA_question_top5_snippets_merged.jsonl.gz",
+        "lines": 582261,
+        "weight": 144
+    },
+    {
+        "name": "yahoo_answers_title_question.jsonl.gz",
+        "lines": 659896,
+        "weight": 163
+    },
+    {
+        "name": "yahoo_answers_question_answer.jsonl.gz",
+        "lines": 681164,
+        "weight": 169
+    },
+    {
+        "name": "yahoo_answers_title_answer.jsonl.gz",
+        "lines": 1198260,
+        "weight": 247
+    },
+    {
+        "name": "amazon-qa-train-pairs.jsonl.gz",
+        "lines": 2448839,
+        "weight": 247
+    },
+    {
+        "name": "gooaq_pairs.jsonl.gz",
+        "lines": 3012496,
+        "weight": 247
+    },
+    {
+        "name": "msmarco-query_passage_negative.jsonl.gz",
+        "lines": 9144553,
+        "weight": 247
+    }
+]

examples/India_900498_S2Hand.tif ADDED Viewed

Git LFS Details

SHA256: ee898621b387a731503a01960397599f209b45c268d4e088689928c704dbe968
Pointer size: 132 Bytes
Size of remote file: 2.15 MB

examples/Spain_7370579_S2Hand.tif ADDED Viewed

Git LFS Details

SHA256: 16e997e6a7159fa11160faf00591da763eb37ac82faf806f7fe733991944a048
Pointer size: 132 Bytes
Size of remote file: 2.32 MB

examples/USA_430764_S2Hand.tif ADDED Viewed

Git LFS Details

SHA256: 385418e105dc1068a7d78585e4395bcdc134e7b0d092ba942867ef74393d8d12
Pointer size: 132 Bytes
Size of remote file: 2.2 MB

examples/subsetted_512x512_HLS.S30.T10SEH.2018190.v1.4_merged.tif ADDED Viewed

Git LFS Details

SHA256: 13bc592a5e569d837bd8bb3524bb0d2f28418830bcc7b0750e74033078f8b17e
Pointer size: 132 Bytes
Size of remote file: 6.3 MB

examples/subsetted_512x512_HLS.S30.T10SFF.2018190.v1.4_merged.tif ADDED Viewed

Git LFS Details

SHA256: b491445bcca5d23a534765ac9f8b24f4cb0c9a75a7254c969456d65a982207a5
Pointer size: 132 Bytes
Size of remote file: 6.3 MB

examples/subsetted_512x512_HLS.S30.T10SGF.2020217.v1.4_merged.tif ADDED Viewed

Git LFS Details

SHA256: ecaa478fdb21ed437ea03436da87ed5efbd3d980e4da23cbee05171212c40378
Pointer size: 132 Bytes
Size of remote file: 6.3 MB

inference.py ADDED Viewed

	@@ -0,0 +1,335 @@

+import argparse
+import os
+from typing import List, Union
+import re
+import datetime
+import numpy as np
+import rasterio
+import torch
+import yaml
+from einops import rearrange
+from terratorch.cli_tools import LightningInferenceModel
+NO_DATA = -9999
+NO_DATA_FLOAT = 0.0001
+OFFSET = 0
+PERCENTILE = 99
+def process_channel_group(orig_img, channels):
+    """
+    Args:
+        orig_img: torch.Tensor representing original image (reference) with shape = (bands, H, W).
+        channels: list of indices representing RGB channels.
+    Returns:
+        torch.Tensor with shape (num_channels, height, width) for original image
+    """
+    orig_img = orig_img[channels, ...]
+    valid_mask = torch.ones_like(orig_img, dtype=torch.bool)
+    valid_mask[orig_img == NO_DATA_FLOAT] = False
+    # Rescale (enhancing contrast)
+    max_value = max(3000, np.percentile(orig_img[valid_mask], PERCENTILE))
+    min_value = OFFSET
+    orig_img = torch.clamp((orig_img - min_value) / (max_value - min_value), 0, 1)
+    # No data as zeros
+    orig_img[~valid_mask] = 0
+    return orig_img
+def read_geotiff(file_path: str):
+    """Read all bands from *file_path* and return image + meta info.
+    Args:
+        file_path: path to image file.
+    Returns:
+        np.ndarray with shape (bands, height, width)
+        meta info dict
+    """
+    with rasterio.open(file_path) as src:
+        img = src.read()
+        meta = src.meta
+        try:
+            coords = src.lnglat()
+        except:
+            # Cannot read coords
+            coords = None
+    return img, meta, coords
+def save_geotiff(image, output_path: str, meta: dict):
+    """Save multi-band image in Geotiff file.
+    Args:
+        image: np.ndarray with shape (bands, height, width)
+        output_path: path where to save the image
+        meta: dict with meta info.
+    """
+    with rasterio.open(output_path, "w", **meta) as dest:
+        for i in range(image.shape[0]):
+            dest.write(image[i, :, :], i + 1)
+    return
+def _convert_np_uint8(float_image: torch.Tensor):
+    image = float_image.numpy() * 255.0
+    image = image.astype(dtype=np.uint8)
+    return image
+def load_example(
+    file_paths: List[str],
+    mean: List[float] = None,
+    std: List[float] = None,
+    indices: Union[list[int], None] = None,
+):
+    """Build an input example by loading images in *file_paths*.
+    Args:
+        file_paths: list of file paths .
+        mean: list containing mean values for each band in the images in *file_paths*.
+        std: list containing std values for each band in the images in *file_paths*.
+    Returns:
+        np.array containing created example
+        list of meta info for each image in *file_paths*
+    """
+    imgs = []
+    metas = []
+    temporal_coords = []
+    location_coords = []
+    for file in file_paths:
+        img, meta, coords = read_geotiff(file)
+        # Rescaling (don't normalize on nodata)
+        img = np.moveaxis(img, 0, -1)  # channels last for rescaling
+        if indices is not None:
+            img = img[..., indices]
+        if mean is not None and std is not None:
+            img = np.where(img == NO_DATA, NO_DATA_FLOAT, (img - mean) / std)
+        imgs.append(img)
+        metas.append(meta)
+        if coords is not None:
+            location_coords.append(coords)
+        try:
+            match = re.search(r'(\d{7,8}T\d{6})', file)
+            if match:
+                year = int(match.group(1)[:4])
+                julian_day = match.group(1).split('T')[0][4:]
+                if len(julian_day) == 3:
+                    julian_day = int(julian_day)
+                else:
+                    julian_day = datetime.datetime.strptime(julian_day, '%m%d').timetuple().tm_yday
+                temporal_coords.append([year, julian_day])
+        except Exception as e:
+            print(f'Could not extract timestamp for {file} ({e})')
+    imgs = np.stack(imgs, axis=0)  # num_frames, H, W, C
+    imgs = np.moveaxis(imgs, -1, 0).astype("float32")  # C, num_frames, H, W
+    imgs = np.expand_dims(imgs, axis=0)  # add batch di
+    return imgs, temporal_coords, location_coords, metas
+def run_model(input_data, model, datamodule, img_size):
+    # Reflect pad if not divisible by img_size
+    original_h, original_w = input_data.shape[-2:]
+    pad_h = (img_size - (original_h % img_size)) % img_size
+    pad_w = (img_size - (original_w % img_size)) % img_size
+    input_data = np.pad(
+        input_data, ((0, 0), (0, 0), (0, 0), (0, pad_h), (0, pad_w)), mode="reflect"
+    )
+    # Build sliding window
+    batch_size = 1
+    batch = torch.tensor(input_data, device="cpu")
+    windows = batch.unfold(3, img_size, img_size).unfold(4, img_size, img_size)
+    h1, w1 = windows.shape[3:5]
+    windows = rearrange(
+        windows, "b c t h1 w1 h w -> (b h1 w1) c t h w", h=img_size, w=img_size
+    )
+    # Split into batches if number of windows > batch_size
+    num_batches = windows.shape[0] // batch_size if windows.shape[0] > batch_size else 1
+    windows = torch.tensor_split(windows, num_batches, dim=0)
+    # Run model
+    pred_imgs = []
+    for x in windows:
+        # Apply standardization
+        x = datamodule.test_transform(image=x.squeeze().numpy().transpose(1,2,0))
+        x['image'] = x['image'].unsqueeze(0)
+        x = datamodule.aug(x)['image']
+        with torch.no_grad():
+            x = x.to(model.device)
+            pred = model(x)
+            pred = pred.output.detach().cpu()
+        y_hat = pred.argmax(dim=1)
+        y_hat = torch.nn.functional.interpolate(y_hat.unsqueeze(1).float(), size=img_size, mode="nearest")
+        pred_imgs.append(y_hat)
+    pred_imgs = torch.concat(pred_imgs, dim=0)
+    # Build images from patches
+    pred_imgs = rearrange(
+        pred_imgs,
+        "(b h1 w1) c h w -> b c (h1 h) (w1 w)",
+        h=img_size,
+        w=img_size,
+        b=1,
+        c=1,
+        h1=h1,
+        w1=w1,
+    )
+    # Cut padded area back to original size
+    pred_imgs = pred_imgs[..., :original_h, :original_w]
+    # Squeeze (batch size 1)
+    pred_imgs = pred_imgs[0]
+    return pred_imgs
+def main(
+    data_file: str,
+    config: str,
+    checkpoint: str,
+    output_dir: str,
+    rgb_outputs: bool,
+    input_indices: list[int] = None,
+):
+    os.makedirs(output_dir, exist_ok=True)
+    with open(config, "r") as f:
+        config_dict = yaml.safe_load(f)
+    # Load model ---------------------------------------------------------------------------------
+    lightning_model = LightningInferenceModel.from_config(config, checkpoint)
+    img_size = 512  # Size of BurnScars
+    # Loading data ---------------------------------------------------------------------------------
+    input_data, temporal_coords, location_coords, meta_data = load_example(
+        file_paths=[data_file], indices=input_indices,
+    )
+    meta_data = meta_data[0]  # only one image
+    if input_data.mean() > 1:
+        input_data = input_data / 10000  # Convert to range 0-1
+    # Running model --------------------------------------------------------------------------------
+    lightning_model.model.eval()
+    channels = config_dict['data']['init_args']['rgb_indices']
+    pred = run_model(input_data, lightning_model.model, lightning_model.datamodule, img_size)
+    # Save pred
+    meta_data.update(count=1, dtype="uint8", compress="lzw", nodata=0)
+    pred_file = os.path.join(output_dir, f"pred_{os.path.splitext(os.path.basename(data_file))[0]}.tiff")
+    save_geotiff(_convert_np_uint8(pred), pred_file, meta_data)
+    # Save image + pred
+    meta_data.update(count=3, dtype="uint8", compress="lzw", nodata=0)
+    if input_data.mean() < 1:
+        input_data = input_data * 10000  # Scale to 0-10000
+    rgb_orig = process_channel_group(
+        orig_img=torch.Tensor(input_data[0, :, 0, ...]),
+        channels=channels,
+    )
+    pred[pred == 0.] = np.nan
+    img_pred = rgb_orig * 0.7 + pred * 0.3
+    img_pred[img_pred.isnan()] = rgb_orig[img_pred.isnan()]
+    img_pred_file = os.path.join(output_dir, f"rgb_pred_{os.path.splitext(os.path.basename(data_file))[0]}.tiff")
+    save_geotiff(
+        image=_convert_np_uint8(img_pred),
+        output_path=img_pred_file,
+        meta=meta_data,
+    )
+    # Save image rgb
+    if rgb_outputs:
+        rgb_file = os.path.join(output_dir, f"original_rgb_{os.path.splitext(os.path.basename(data_file))[0]}.tiff")
+        save_geotiff(
+            image=_convert_np_uint8(rgb_orig),
+            output_path=rgb_file,
+            meta=meta_data,
+        )
+    print("Done!")
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser("run inference", add_help=False)
+    parser.add_argument(
+        "--data_file",
+        type=str,
+        default="examples/subsetted_512x512_HLS.S30.T10SEH.2018190.v1.4_merged.tif",
+        help="Path to the file.",
+    )
+    parser.add_argument(
+        "--config",
+        "-c",
+        type=str,
+        default="burn_scars_config.yaml",
+        help="Path to yaml file containing model parameters.",
+    )
+    parser.add_argument(
+        "--checkpoint",
+        type=str,
+        default="Prithvi_EO_V2_300M_BurnScars.pt",
+        help="Path to a checkpoint file to load from.",
+    )
+    parser.add_argument(
+        "--output_dir",
+        type=str,
+        default="output",
+        help="Path to the directory where to save outputs.",
+    )
+    parser.add_argument(
+        "--input_indices",
+        default=[0,1,2,3,4,5],
+        type=int,
+        nargs="+",
+        help="0-based indices of the six Prithvi channels to be selected from the input. By default selects [0,1,2,3,4,5] for filtered HLS data.",
+    )
+    parser.add_argument(
+        "--rgb_outputs",
+        action="store_true",
+        help="If present, output files will only contain RGB channels. "
+        "Otherwise, all bands will be saved.",
+    )
+    args = parser.parse_args()
+    main(**vars(args))

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:53aa51172d142c89d9012cce15ae4d6cc0ca6895895114379cacb4fab128d9db
+size 90868376

modules.json ADDED Viewed

	@@ -0,0 +1,20 @@

+[
+  {
+    "idx": 0,
+    "name": "0",
+    "path": "",
+    "type": "sentence_transformers.models.Transformer"
+  },
+  {
+    "idx": 1,
+    "name": "1",
+    "path": "1_Pooling",
+    "type": "sentence_transformers.models.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.models.Normalize"
+  }
+]

onnx/model.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6fd5d72fe4589f189f8ebc006442dbb529bb7ce38f8082112682524616046452
+size 90405214

onnx/model_O1.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1391c6fc20b5530250bc15cbe1f47578ffeca55ab0551d335cc668b6299a88ec
+size 90360328

onnx/model_O2.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1de3905029190b398c7d300b530e320cf4b5e7d3dfb9af1429ebd73fd9a16faf
+size 90326566

onnx/model_O3.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a44f671e364dddbac31f203f07b91be6b0a35e51936e5ebfab65b6d9538b83ff
+size 90326497

onnx/model_O4.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1667d7f3ba669048b13a96ee3a44456d5e42c8f44588ae8b603430e16160c485
+size 45212349

onnx/model_qint8_arm64.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4278337fd0ff3c68bfb6291042cad8ab363e1d9fbc43dcb499fe91c871902474
+size 23026053

onnx/model_qint8_avx512.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4278337fd0ff3c68bfb6291042cad8ab363e1d9fbc43dcb499fe91c871902474
+size 23026053

onnx/model_qint8_avx512_vnni.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4278337fd0ff3c68bfb6291042cad8ab363e1d9fbc43dcb499fe91c871902474
+size 23026053

onnx/model_quint8_avx2.onnx ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b941bf19f1f1283680f449fa6a7336bb5600bdcd5f84d10ddc5cd72218a0fd21
+size 23046789

openvino/openvino_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8b86cab4722e2aefab310cf96d4d5a9eb3b187f7d9670a082afc55c7fa0d392a
+size 90265744

openvino/openvino_model.xml ADDED Viewed

The diff for this file is too large to render. See raw diff

openvino/openvino_model_qint8_quantized.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c92ea4af3c6bc7b4a0f3b3d61b147c850f4dbdd7c9e7beee0c0c70dc12da289b
+size 22933664

openvino/openvino_model_qint8_quantized.xml ADDED Viewed

The diff for this file is too large to render. See raw diff

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c3a85f238711653950f6a79ece63eb0ea93d76f6a6284be04019c53733baf256
+size 90888945

requirements.txt ADDED Viewed

	@@ -0,0 +1,6 @@

+torch
+torchvision
+timm
+einops
+rasterio
+terratorch==0.99.8

rust_model.ot ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2d98d96d278348988f2744e6445b8bc16d921c3f6e17c667362f3cb353007aea
+size 90887379

sentence_bert_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "max_seq_length": 256,
+  "do_lower_case": false
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]"}

splits/test.txt ADDED Viewed

	@@ -0,0 +1,120 @@

+T10SDH.2020248.v1
+T10SEH.2018190.v1
+T10SEH.2018245.v1
+T10SEH.2018280.v1
+T10SEH.2019305.v1
+T10SEH.2020190.v1
+T10SEH.2020285.v1
+T10SEJ.2019185.v1
+T10TFQ.2018183.v1
+T10TFQ.2018245.v1
+T10TFT.2018213.v1
+T10TGN.2019305.v1
+T10TGN.2020310.v1
+T10TGR.2020275.v1
+T10TGS.2018245.v1
+T10TGS.2018285.v1
+T10TGS.2019195.v1
+T10TGS.2020215.v1
+T10TGT.2018188.v1
+T10TGT.2018213.v1
+T10TGT.2018285.v1
+T10TGT.2019213.v1
+T10TGT.2020218.v1
+T10UGU.2018213.v1
+T10UGU.2020215.v1
+T10UGU.2020280.v1
+T10UGU.2021249.v1
+T11SLB.2018197.v1
+T11SQC.2020196.v1
+T11TLE.2018247.v1
+T11TLH.2019215.v1
+T11TMG.2019217.v1
+T11TNJ.2019217.v1
+T11TPH.2018244.v1
+T11TPH.2020174.v1
+T11TPH.2021263.v1
+T11TPL.2021223.v1
+T11TQH.2018219.v1
+T11TQH.2019244.v1
+T11UQP.2018249.v1
+T12RWV.2019075.v1
+T12RWV.2019225.v1
+T12SUC.2019223.v1
+T12SUC.2020153.v1
+T12SUC.2020248.v1
+T12SUC.2020318.v1
+T12SUJ.2019298.v1
+T12SVC.2018215.v1
+T12SVC.2019245.v1
+T12SVC.2019280.v1
+T12SVC.2020155.v1
+T12SVC.2020190.v1
+T12SVC.2020250.v1
+T12SVC.2020285.v1
+T12SVD.2019183.v1
+T12SVD.2020218.v1
+T12SVE.2019183.v1
+T12SVE.2019228.v1
+T12SWA.2018225.v1
+T12SXA.2018157.v1
+T12SXA.2020187.v1
+T12SYG.2020220.v1
+T12TUK.2020286.v1
+T12TVP.2018221.v1
+T12TXK.2018215.v1
+T12TXT.2018293.v1
+T12TXT.2020248.v1
+T13REP.2018141.v1
+T13REQ.2018156.v1
+T13SBS.2020217.v1
+T13SDV.2020269.v1
+T13SEA.2018144.v1
+T13SFC.2020184.v1
+T13TCG.2020277.v1
+T13TCH.2020280.v1
+T13TCM.2020250.v1
+T13TCN.2020278.v1
+T13TDH.2018292.v1
+T13TDL.2019150.v1
+T13TDL.2020245.v1
+T13TDL.2020280.v1
+T13TDM.2020250.v1
+T14SMC.2018213.v1
+T14SME.2018138.v1
+T14SMF.2018098.v1
+T14SPB.2018035.v1
+T14SPF.2019345.v1
+T14SQE.2018075.v1
+T14SQE.2020065.v1
+T14SQF.2018125.v1
+T15RVQ.2018094.v1
+T15RVQ.2019099.v1
+T15STV.2018102.v1
+T15SXB.2018154.v1
+T15SXB.2019134.v1
+T15SXB.2020089.v1
+T15SXB.2020099.v1
+T15SXB.2021093.v1
+T16SBA.2019206.v1
+T16SBD.2020096.v1
+T16SCF.2019111.v1
+T16SGG.2021094.v1
+T16TFS.2019153.v1
+T17RML.2018064.v1
+T17RML.2021103.v1
+T17RNL.2019111.v1
+T17SKT.2018107.v1
+T17SKT.2018132.v1
+T17SKV.2019100.v1
+T17SKV.2021094.v1
+T17SLV.2019112.v1
+T17SMS.2019094.v1
+T17SMS.2021103.v1
+T17SMU.2018132.v1
+T17SNS.2021128.v1
+T17SPS.2018074.v1
+T17SPS.2019009.v1
+T17SPS.2019094.v1
+T17SPS.2020039.v1
+T18SUD.2020061.v1

splits/train.txt ADDED Viewed

	@@ -0,0 +1,524 @@

+T10SEJ.2018185.v1
+T10SFE.2020267.v1
+T10SFE.2021166.v1
+T10SFF.2018155.v1
+T10SFF.2018190.v1
+T10SFF.2020215.v1
+T10SFF.2020250.v1
+T10SFF.2021189.v1
+T10SFG.2020215.v1
+T10SFH.2018185.v1
+T10SFH.2020185.v1
+T10SFH.2020245.v1
+T10SGD.2018257.v1
+T10SGD.2021306.v1
+T10SGE.2018247.v1
+T10SGE.2019187.v1
+T10SGE.2020162.v1
+T10SGE.2020187.v1
+T10SGE.2020217.v1
+T10SGE.2020247.v1
+T10SGF.2020217.v1
+T10SGG.2018187.v1
+T10SGG.2019307.v1
+T10SGG.2020247.v1
+T10TDN.2019213.v1
+T10TEK.2018183.v1
+T10TEK.2018340.v1
+T10TEK.2020275.v1
+T10TEM.2018213.v1
+T10TEN.2019168.v1
+T10TFK.2020220.v1
+T10TFL.2018215.v1
+T10TFL.2018245.v1
+T10TFL.2020215.v1
+T10TFN.2018175.v1
+T10TFN.2018245.v1
+T10TFN.2020235.v1
+T10TFP.2018285.v1
+T10TFP.2019278.v1
+T10TFP.2020248.v1
+T10TFQ.2018173.v1
+T10TFQ.2019245.v1
+T10TFQ.2019305.v1
+T10TFR.2018188.v1
+T10TFR.2018213.v1
+T10TFR.2020173.v1
+T10TFS.2018193.v1
+T10TFS.2018213.v1
+T10TFS.2019213.v1
+T10TGK.2019245.v1
+T10TGK.2019280.v1
+T10TGK.2020285.v1
+T10TGL.2018215.v1
+T10TGL.2019245.v1
+T10TGL.2020265.v1
+T10TGP.2018215.v1
+T10TGQ.2018245.v1
+T10TGQ.2020275.v1
+T10TGR.2018215.v1
+T10TGR.2018245.v1
+T10TGR.2019215.v1
+T10TGS.2018190.v1
+T10TGS.2018215.v1
+T10TGS.2019215.v1
+T10TGS.2020245.v1
+T10UGU.2018245.v1
+T10UGV.2020218.v1
+T11SKB.2018222.v1
+T11SKB.2020222.v1
+T11SKU.2019002.v1
+T11SKV.2019152.v1
+T11SKV.2019187.v1
+T11SLB.2019237.v1
+T11SLC.2020247.v1
+T11SLT.2018349.v1
+T11SLT.2019309.v1
+T11SLT.2021163.v1
+T11SLU.2018184.v1
+T11SLU.2018274.v1
+T11SLU.2020229.v1
+T11SLU.2020249.v1
+T11SLV.2018249.v1
+T11SLV.2020222.v1
+T11SLV.2020247.v1
+T11SLV.2021186.v1
+T11SLV.2021216.v1
+T11SLV.2021251.v1
+T11SLV.2021331.v1
+T11SMS.2018259.v1
+T11SMS.2020029.v1
+T11SMT.2018154.v1
+T11SMT.2018249.v1
+T11SMT.2019294.v1
+T11SMT.2019309.v1
+T11SMT.2020194.v1
+T11SMT.2020249.v1
+T11SMT.2020289.v1
+T11SMT.2020309.v1
+T11SMT.2021248.v1
+T11SMU.2020299.v1
+T11SMV.2020234.v1
+T11SMV.2020249.v1
+T11SNS.2019246.v1
+T11SNS.2020276.v1
+T11SNS.2021155.v1
+T11SNT.2018216.v1
+T11SNT.2020221.v1
+T11SNT.2020281.v1
+T11SPA.2020196.v1
+T11SPB.2019281.v1
+T11SPB.2020196.v1
+T11SPB.2020241.v1
+T11SPV.2020186.v1
+T11SQA.2019226.v1
+T11SQB.2020241.v1
+T11SQC.2020241.v1
+T11SQD.2020196.v1
+T11TKE.2018215.v1
+T11TKF.2020265.v1
+T11TKG.2020265.v1
+T11TLE.2018182.v1
+T11TLE.2019257.v1
+T11TLG.2018247.v1
+T11TLH.2018152.v1
+T11TLH.2018217.v1
+T11TLH.2020247.v1
+T11TLJ.2019155.v1
+T11TLJ.2019327.v1
+T11TLM.2018245.v1
+T11TLM.2019245.v1
+T11TLM.2019305.v1
+T11TLM.2020275.v1
+T11TLN.2020280.v1
+T11TMF.2019312.v1
+T11TMF.2020217.v1
+T11TMG.2018222.v1
+T11TMH.2018182.v1
+T11TMH.2019227.v1
+T11TMH.2020247.v1
+T11TMJ.2018217.v1
+T11TMJ.2020247.v1
+T11TMK.2018217.v1
+T11TMK.2018292.v1
+T11TMK.2020247.v1
+T11TMM.2018285.v1
+T11TMM.2020245.v1
+T11TMM.2021224.v1
+T11TMN.2018245.v1
+T11TNE.2019224.v1
+T11TNF.2018199.v1
+T11TNF.2018219.v1
+T11TNF.2018244.v1
+T11TNF.2018289.v1
+T11TNF.2019224.v1
+T11TNF.2019314.v1
+T11TNH.2018217.v1
+T11TNH.2020217.v1
+T11TNJ.2018217.v1
+T11TNJ.2019244.v1
+T11TNK.2018222.v1
+T11TPE.2018244.v1
+T11TPE.2019269.v1
+T11TPE.2020219.v1
+T11TPF.2018219.v1
+T11TPF.2018289.v1
+T11TPF.2021183.v1
+T11TPH.2018219.v1
+T11TPH.2019244.v1
+T11TPH.2020184.v1
+T11TPH.2020219.v1
+T11TPH.2020249.v1
+T11TPH.2021238.v1
+T11TPJ.2020249.v1
+T11TPL.2019244.v1
+T11TPM.2021223.v1
+T11TPN.2019214.v1
+T11TPN.2020217.v1
+T11TQG.2018221.v1
+T11TQG.2018291.v1
+T11TQG.2020306.v1
+T11TQH.2019221.v1
+T11TQH.2020216.v1
+T11TQH.2020251.v1
+T11TQJ.2018219.v1
+T11TQK.2020249.v1
+T11ULP.2018245.v1
+T11ULP.2020215.v1
+T12RVV.2018215.v1
+T12RVV.2020320.v1
+T12RXV.2018182.v1
+T12RXV.2019062.v1
+T12RYV.2018152.v1
+T12STC.2020218.v1
+T12STE.2020246.v1
+T12STF.2018231.v1
+T12STF.2018291.v1
+T12STF.2021190.v1
+T12STF.2021215.v1
+T12STG.2018186.v1
+T12STH.2020246.v1
+T12SUC.2019158.v1
+T12SUD.2018168.v1
+T12SUD.2018218.v1
+T12SUD.2019183.v1
+T12SUD.2020218.v1
+T12SUE.2020218.v1
+T12SUF.2018253.v1
+T12SUH.2018228.v1
+T12SUH.2019298.v1
+T12SUH.2020268.v1
+T12SVA.2020310.v1
+T12SVB.2020155.v1
+T12SVB.2020185.v1
+T12SVB.2020310.v1
+T12SVC.2019190.v1
+T12SVF.2018253.v1
+T12SWA.2019260.v1
+T12SWA.2020230.v1
+T12SWB.2019155.v1
+T12SWB.2020155.v1
+T12SWB.2020250.v1
+T12SWC.2019225.v1
+T12SWC.2020190.v1
+T12SWC.2020250.v1
+T12SXB.2018217.v1
+T12SXG.2019235.v1
+T12SXJ.2020200.v1
+T12SYG.2018225.v1
+T12TTM.2018219.v1
+T12TTM.2019244.v1
+T12TTM.2020306.v1
+T12TUK.2018231.v1
+T12TUM.2019261.v1
+T12TUM.2020191.v1
+T12TUN.2018186.v1
+T12TUN.2018216.v1
+T12TUN.2018246.v1
+T12TUN.2019276.v1
+T12TUN.2020216.v1
+T12TUN.2021150.v1
+T12TUN.2021205.v1
+T12TUN.2021215.v1
+T12TVK.2020188.v1
+T12TVK.2020228.v1
+T12TVK.2020308.v1
+T12TVM.2018246.v1
+T12TVN.2018246.v1
+T12TVN.2018291.v1
+T12TVR.2018246.v1
+T12TVS.2019216.v1
+T12TVS.2020191.v1
+T12TVT.2018196.v1
+T12TWK.2020268.v1
+T12TWT.2020276.v1
+T12TXL.2018220.v1
+T12TXL.2020250.v1
+T12TXQ.2018248.v1
+T12TXR.2020283.v1
+T12TYK.2018220.v1
+T12TYK.2020215.v1
+T12TYL.2018290.v1
+T12TYP.2018220.v1
+T12TYP.2018245.v1
+T12TYP.2020215.v1
+T12TYQ.2018185.v1
+T12TYT.2018153.v1
+T12TYT.2018213.v1
+T12TYT.2019153.v1
+T12TYT.2019248.v1
+T12UXU.2021250.v1
+T13RGP.2020118.v1
+T13SBT.2019202.v1
+T13SBT.2019237.v1
+T13SBV.2019247.v1
+T13SCA.2019227.v1
+T13SCR.2018214.v1
+T13SDA.2018199.v1
+T13SDS.2018214.v1
+T13SDT.2018134.v1
+T13SDT.2019184.v1
+T13SDT.2020249.v1
+T13SEA.2020214.v1
+T13SEB.2018189.v1
+T13SEB.2020249.v1
+T13SEC.2018164.v1
+T13SER.2018156.v1
+T13SEV.2020154.v1
+T13SFA.2018134.v1
+T13SFA.2018159.v1
+T13SFA.2018184.v1
+T13SFA.2019244.v1
+T13SFA.2020154.v1
+T13SFB.2019109.v1
+T13SFB.2019219.v1
+T13SFB.2020164.v1
+T13SFB.2020189.v1
+T13SFC.2019154.v1
+T13SFC.2020124.v1
+T13SFR.2020216.v1
+T13SFS.2018136.v1
+T13SFS.2018196.v1
+T13SFT.2018136.v1
+T13SFT.2019096.v1
+T13SFT.2019171.v1
+T13SGB.2018136.v1
+T13SGB.2019076.v1
+T13TCG.2020247.v1
+T13TCJ.2020200.v1
+T13TCJ.2020230.v1
+T13TCK.2020215.v1
+T13TCK.2020245.v1
+T13TCL.2020245.v1
+T13TCL.2020280.v1
+T13TCM.2020215.v1
+T13TDF.2018222.v1
+T13TDK.2020217.v1
+T13TDK.2020280.v1
+T13TDL.2020187.v1
+T13TDL.2020307.v1
+T13TEE.2020214.v1
+T13TEE.2020249.v1
+T13TEF.2019109.v1
+T13TEG.2019157.v1
+T13TEL.2020307.v1
+T13TEN.2018245.v1
+T13TFJ.2018319.v1
+T13TFN.2018157.v1
+T13TFN.2019152.v1
+T13TGE.2018136.v1
+T13TGH.2020304.v1
+T13UCP.2020248.v1
+T13UDP.2018290.v1
+T13UDP.2020255.v1
+T13UEP.2018135.v1
+T14RKU.2020158.v1
+T14RKV.2018143.v1
+T14RKV.2018193.v1
+T14RKV.2018213.v1
+T14RKV.2020268.v1
+T14RKV.2020278.v1
+T14RLT.2020273.v1
+T14RLU.2019293.v1
+T14RLU.2020278.v1
+T14RLV.2018043.v1
+T14RLV.2019223.v1
+T14RLV.2019348.v1
+T14RMV.2020220.v1
+T14RNQ.2021051.v1
+T14RNU.2018215.v1
+T14RNV.2018215.v1
+T14RNV.2020220.v1
+T14RPQ.2018107.v1
+T14RPR.2020187.v1
+T14RPS.2019257.v1
+T14RQS.2019322.v1
+T14RQS.2020032.v1
+T14SKF.2019246.v1
+T14SLA.2020223.v1
+T14SLC.2018038.v1
+T14SLD.2020218.v1
+T14SLE.2018138.v1
+T14SLE.2019098.v1
+T14SLE.2019248.v1
+T14SLE.2020323.v1
+T14SLF.2020108.v1
+T14SMA.2020185.v1
+T14SMB.2019228.v1
+T14SMB.2020268.v1
+T14SMD.2018138.v1
+T14SNA.2018215.v1
+T14SNB.2018215.v1
+T14SNB.2019280.v1
+T14SNG.2020183.v1
+T14SPD.2018100.v1
+T14SQE.2018100.v1
+T14SQF.2018075.v1
+T14SQF.2018100.v1
+T14SQG.2018125.v1
+T14SQJ.2020340.v1
+T14TKM.2019144.v1
+T14TQT.2021112.v1
+T14UPU.2021112.v1
+T14UQU.2019163.v1
+T15RTN.2020032.v1
+T15RTQ.2021101.v1
+T15RUQ.2018094.v1
+T15RUQ.2018129.v1
+T15RUQ.2019099.v1
+T15RUQ.2021088.v1
+T15RVP.2021063.v1
+T15RVQ.2021063.v1
+T15RWQ.2021063.v1
+T15RWQ.2021098.v1
+T15RXQ.2020106.v1
+T15RXQ.2021095.v1
+T15RYQ.2021095.v1
+T15STA.2018125.v1
+T15STC.2018125.v1
+T15STD.2018125.v1
+T15STU.2018237.v1
+T15SUB.2018127.v1
+T15SUC.2018127.v1
+T15SUU.2018062.v1
+T15SUU.2020107.v1
+T15SUV.2018102.v1
+T15SUV.2020107.v1
+T15SVA.2018102.v1
+T15SVU.2018114.v1
+T15SVU.2019144.v1
+T15SVU.2021063.v1
+T15SVU.2021098.v1
+T15SWA.2021093.v1
+T15SWS.2021128.v1
+T15SWV.2018099.v1
+T15SWV.2019099.v1
+T15SWV.2021093.v1
+T15SXA.2018154.v1
+T15SXA.2021093.v1
+T15SXB.2019099.v1
+T15SXC.2019099.v1
+T15SYR.2021110.v1
+T15TVN.2018135.v1
+T15TVN.2019140.v1
+T15TXM.2018187.v1
+T15TXM.2019157.v1
+T16RBU.2018133.v1
+T16RBU.2019133.v1
+T16RBU.2020123.v1
+T16RBV.2018158.v1
+T16RBV.2020033.v1
+T16RBV.2021127.v1
+T16RCU.2019133.v1
+T16RCU.2020033.v1
+T16RCU.2020118.v1
+T16RCU.2021077.v1
+T16RCV.2020118.v1
+T16REV.2018095.v1
+T16REV.2018130.v1
+T16REV.2019080.v1
+T16REV.2019100.v1
+T16REV.2019135.v1
+T16REV.2019165.v1
+T16REV.2020060.v1
+T16REV.2020100.v1
+T16REV.2021129.v1
+T16REV.2021314.v1
+T16RFT.2019250.v1
+T16RFU.2019105.v1
+T16RFU.2019250.v1
+T16RFU.2019305.v1
+T16RFU.2021054.v1
+T16RFU.2021069.v1
+T16RFU.2021109.v1
+T16RGU.2018107.v1
+T16RGU.2019112.v1
+T16RGU.2019127.v1
+T16RGU.2019267.v1
+T16RGU.2019277.v1
+T16RGU.2020107.v1
+T16RGU.2021066.v1
+T16RGU.2021096.v1
+T16RGU.2021206.v1
+T16SCA.2018133.v1
+T16SCG.2019111.v1
+T16SDB.2018063.v1
+T16SDB.2018098.v1
+T16SDB.2018133.v1
+T16SDB.2021062.v1
+T16SDC.2018098.v1
+T16SDC.2020118.v1
+T16SDC.2021127.v1
+T16SDD.2020093.v1
+T16SEB.2018090.v1
+T16SEB.2018095.v1
+T16SEB.2018155.v1
+T16SEB.2019100.v1
+T16SEB.2019135.v1
+T16SEB.2021069.v1
+T16SEC.2020105.v1
+T16SEC.2021094.v1
+T16SFB.2019100.v1
+T16SFB.2021094.v1
+T16SFC.2018155.v1
+T16SFC.2019100.v1
+T16SFC.2019165.v1
+T16SFC.2020045.v1
+T16SFC.2020105.v1
+T16SFC.2021094.v1
+T16SFD.2019100.v1
+T16SFF.2021094.v1
+T16SGD.2018155.v1
+T16SGD.2019100.v1
+T16SGE.2019100.v1
+T16TDS.2019159.v1
+T16TGQ.2018133.v1
+T16TGQ.2019158.v1
+T16TGQ.2021167.v1
+T17RKN.2018087.v1
+T17RKN.2019112.v1
+T17RMH.2021103.v1
+T17RMJ.2018064.v1
+T17RMJ.2021313.v1
+T17RMN.2020034.v1
+T17RMN.2020094.v1
+T17RMN.2021063.v1
+T17RNK.2019081.v1
+T17RNK.2020126.v1
+T17SKS.2021126.v1
+T17SKU.2018107.v1
+T17SKU.2021094.v1
+T17SLB.2019287.v1
+T17SLT.2019112.v1
+T17SLT.2021096.v1
+T17SMA.2019112.v1
+T17SMA.2021096.v1
+T17SMV.2021096.v1
+T17SNA.2021348.v1
+T17SNB.2019162.v1
+T17SNB.2020102.v1
+T17SNU.2018074.v1
+T17SNU.2018129.v1
+T17SQU.2018121.v1
+T18SVJ.2020131.v1
+T18TXQ.2018266.v1

splits/val.txt ADDED Viewed

	@@ -0,0 +1,160 @@

+T10SEJ.2018220.v1
+T10SFG.2020185.v1
+T10TEK.2019275.v1
+T10TEK.2019350.v1
+T10TFM.2018110.v1
+T10TFM.2018155.v1
+T10TFM.2019215.v1
+T10TFM.2019280.v1
+T10TFM.2020215.v1
+T10TGM.2020215.v1
+T10TGR.2019245.v1
+T11SKD.2018192.v1
+T11SKD.2020197.v1
+T11SKD.2020217.v1
+T11SLU.2021188.v1
+T11SNB.2018224.v1
+T11SNB.2020234.v1
+T11SNB.2021153.v1
+T11SNB.2021268.v1
+T11SPD.2020184.v1
+T11SPV.2020236.v1
+T11SPV.2020246.v1
+T11SPV.2021215.v1
+T11SQA.2020286.v1
+T11SQS.2019078.v1
+T11TLL.2018190.v1
+T11TLL.2018215.v1
+T11TLL.2018245.v1
+T11TLL.2020275.v1
+T11TME.2019222.v1
+T11TMF.2018222.v1
+T11TMF.2019227.v1
+T11TMF.2019257.v1
+T11TNE.2018199.v1
+T11TNG.2018219.v1
+T11TNG.2018289.v1
+T11TPG.2018219.v1
+T11TPG.2019214.v1
+T11TPG.2020249.v1
+T11TPK.2021268.v1
+T11TQG.2020186.v1
+T11TQG.2020216.v1
+T11TQL.2021223.v1
+T11ULP.2019245.v1
+T11ULP.2020280.v1
+T11ULP.2021249.v1
+T12RXV.2018217.v1
+T12STD.2018168.v1
+T12STD.2020248.v1
+T12STF.2020156.v1
+T12STG.2020241.v1
+T12STG.2020291.v1
+T12STJ.2020241.v1
+T12SUE.2019183.v1
+T12SUF.2019183.v1
+T12SUF.2020183.v1
+T12SUF.2020223.v1
+T12SUH.2020188.v1
+T12SUJ.2019308.v1
+T12SUJ.2020186.v1
+T12SVA.2020230.v1
+T12SVD.2018168.v1
+T12SVD.2018310.v1
+T12SWD.2018220.v1
+T12SWJ.2018238.v1
+T12SXA.2020247.v1
+T12TUM.2018196.v1
+T12TUM.2018246.v1
+T12TUM.2019231.v1
+T12TUP.2019226.v1
+T12TUP.2020241.v1
+T12TVR.2020256.v1
+T12TVR.2020281.v1
+T12TVS.2020256.v1
+T12TWQ.2020283.v1
+T12TWR.2020283.v1
+T12TXM.2018220.v1
+T12TXS.2020278.v1
+T12TXT.2018248.v1
+T12TYS.2020248.v1
+T12TYS.2020278.v1
+T13REP.2019241.v1
+T13SDU.2018219.v1
+T13SDU.2018254.v1
+T13SFR.2020251.v1
+T13TBE.2018220.v1
+T13TBF.2018190.v1
+T13TCL.2019150.v1
+T13TDE.2020247.v1
+T13TFG.2020274.v1
+T13TFH.2020204.v1
+T13TFJ.2020264.v1
+T13TGJ.2020309.v1
+T14RLU.2020193.v1
+T14RMV.2020280.v1
+T14SKA.2019278.v1
+T14SKA.2020223.v1
+T14SKB.2019268.v1
+T14SKD.2018156.v1
+T14SKE.2018156.v1
+T14SKE.2019111.v1
+T14SKE.2020216.v1
+T14SKE.2020281.v1
+T14SLA.2020193.v1
+T14SLC.2019258.v1
+T14SLD.2018138.v1
+T14SLD.2018163.v1
+T14SLD.2018218.v1
+T14SLE.2018098.v1
+T14SMB.2018073.v1
+T14SMB.2018138.v1
+T14SMC.2019258.v1
+T14SND.2018125.v1
+T14SND.2018215.v1
+T14SNF.2018095.v1
+T14SNF.2020183.v1
+T14SNG.2020118.v1
+T14SPH.2018095.v1
+T14SPH.2018125.v1
+T14SPJ.2018125.v1
+T14SPJ.2019345.v1
+T14SQG.2020085.v1
+T14SQH.2018125.v1
+T14UPV.2018136.v1
+T14UQU.2021112.v1
+T15RTM.2020059.v1
+T15RVQ.2020059.v1
+T15RWQ.2019134.v1
+T15SVA.2019092.v1
+T15SVU.2020109.v1
+T15SWA.2019099.v1
+T15SWA.2020109.v1
+T15SWR.2018129.v1
+T15SWR.2021128.v1
+T15TYJ.2018124.v1
+T15TYL.2019159.v1
+T16RCV.2021167.v1
+T16SBA.2018106.v1
+T16SBB.2019111.v1
+T16SBB.2021145.v1
+T16SEH.2021102.v1
+T16SFD.2018100.v1
+T16SGA.2018107.v1
+T16SGF.2018155.v1
+T17RLP.2018082.v1
+T17RLP.2018102.v1
+T17RLP.2018127.v1
+T17RLP.2019112.v1
+T17RLP.2021111.v1
+T17RNM.2019349.v1
+T17RNM.2021103.v1
+T17SKU.2021166.v1
+T17SLS.2020097.v1
+T17SLT.2018107.v1
+T17SLT.2020097.v1
+T17SMR.2021128.v1
+T17SNC.2019112.v1
+T17SQD.2020094.v1
+T18TWK.2018121.v1
+T18TWK.2019093.v1

tf_model.h5 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:24c06a7429b843d46e40c6b167122053921bf94dce2e5550ea5c07fabc597646
+size 91005696

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"do_lower_case": true, "unk_token": "[UNK]", "sep_token": "[SEP]", "pad_token": "[PAD]", "cls_token": "[CLS]", "mask_token": "[MASK]", "tokenize_chinese_chars": true, "strip_accents": null, "name_or_path": "nreimers/MiniLM-L6-H384-uncased", "do_basic_tokenize": true, "never_split": null, "tokenizer_class": "BertTokenizer", "model_max_length": 512}

train_script.py ADDED Viewed

	@@ -0,0 +1,344 @@

+"""
+Train script for a single file
+Need to set the TPU address first:
+export XRT_TPU_CONFIG="localservice;0;localhost:51011"
+"""
+import torch.multiprocessing as mp
+import threading
+import time
+import random
+import sys
+import argparse
+import gzip
+import json
+import logging
+import tqdm
+import torch
+from torch import nn
+from torch.utils.data import DataLoader
+import torch
+import torch_xla
+import torch_xla.core
+import torch_xla.core.functions
+import torch_xla.core.xla_model as xm
+import torch_xla.distributed.xla_multiprocessing as xmp
+import torch_xla.distributed.parallel_loader as pl
+import os
+from shutil import copyfile
+from transformers import (
+    AdamW,
+    AutoModel,
+    AutoTokenizer,
+    get_linear_schedule_with_warmup,
+    set_seed,
+)
+class AutoModelForSentenceEmbedding(nn.Module):
+    def __init__(self, model_name, tokenizer, normalize=True):
+        super(AutoModelForSentenceEmbedding, self).__init__()
+        self.model = AutoModel.from_pretrained(model_name)
+        self.normalize = normalize
+        self.tokenizer = tokenizer
+    def forward(self, **kwargs):
+        model_output = self.model(**kwargs)
+        embeddings = self.mean_pooling(model_output, kwargs['attention_mask'])
+        if self.normalize:
+            embeddings = torch.nn.functional.normalize(embeddings, p=2, dim=1)
+        return embeddings
+    def mean_pooling(self, model_output, attention_mask):
+        token_embeddings = model_output[0]  # First element of model_output contains all token embeddings
+        input_mask_expanded = attention_mask.unsqueeze(-1).expand(token_embeddings.size()).float()
+        return torch.sum(token_embeddings * input_mask_expanded, 1) / torch.clamp(input_mask_expanded.sum(1), min=1e-9)
+    def save_pretrained(self, output_path):
+        if xm.is_master_ordinal():
+            self.tokenizer.save_pretrained(output_path)
+            self.model.config.save_pretrained(output_path)
+        xm.save(self.model.state_dict(), os.path.join(output_path, "pytorch_model.bin"))
+def train_function(index, args, queue):
+    tokenizer = AutoTokenizer.from_pretrained(args.model)
+    model = AutoModelForSentenceEmbedding(args.model, tokenizer)
+    ### Train Loop
+    device = xm.xla_device()
+    model = model.to(device)
+    # Instantiate optimizer
+    optimizer = AdamW(params=model.parameters(), lr=2e-5, correct_bias=True)
+    lr_scheduler = get_linear_schedule_with_warmup(
+        optimizer=optimizer,
+        num_warmup_steps=500,
+        num_training_steps=args.steps,
+    )
+    # Now we train the model
+    cross_entropy_loss = nn.CrossEntropyLoss()
+    max_grad_norm = 1
+    model.train()
+    for global_step in tqdm.trange(args.steps, disable=not xm.is_master_ordinal()):
+        #### Get the batch data
+        batch = queue.get()
+        #print(index, "batch {}x{}".format(len(batch), ",".join([str(len(b)) for b in batch])))
+        if len(batch[0]) == 2: #(anchor, positive)
+            text1 = tokenizer([b[0] for b in batch], return_tensors="pt", max_length=args.max_length, truncation=True, padding="max_length")
+            text2 = tokenizer([b[1] for b in batch], return_tensors="pt", max_length=args.max_length, truncation=True, padding="max_length")
+            ### Compute embeddings
+            embeddings_a = model(**text1.to(device))
+            embeddings_b = model(**text2.to(device))
+            ### Gather all embedings
+            embeddings_a = torch_xla.core.functions.all_gather(embeddings_a)
+            embeddings_b = torch_xla.core.functions.all_gather(embeddings_b)
+            ### Compute similarity scores 512 x 512
+            scores = torch.mm(embeddings_a, embeddings_b.transpose(0, 1)) * args.scale
+            ### Compute cross-entropy loss
+            labels = torch.tensor(range(len(scores)), dtype=torch.long, device=embeddings_a.device)  # Example a[i] should match with b[i]
+            ## Symmetric loss as in CLIP
+            loss = (cross_entropy_loss(scores, labels) + cross_entropy_loss(scores.transpose(0, 1), labels)) / 2
+        else:   #(anchor, positive, negative)
+            text1 = tokenizer([b[0] for b in batch], return_tensors="pt", max_length=args.max_length, truncation=True, padding="max_length")
+            text2 = tokenizer([b[1] for b in batch], return_tensors="pt", max_length=args.max_length, truncation=True, padding="max_length")
+            text3 = tokenizer([b[2] for b in batch], return_tensors="pt", max_length=args.max_length, truncation=True, padding="max_length")
+            embeddings_a  = model(**text1.to(device))
+            embeddings_b1 = model(**text2.to(device))
+            embeddings_b2 = model(**text3.to(device))
+            embeddings_a  = torch_xla.core.functions.all_gather(embeddings_a)
+            embeddings_b1 = torch_xla.core.functions.all_gather(embeddings_b1)
+            embeddings_b2 = torch_xla.core.functions.all_gather(embeddings_b2)
+            embeddings_b = torch.cat([embeddings_b1, embeddings_b2])
+            ### Compute similarity scores 512 x 1024
+            scores = torch.mm(embeddings_a, embeddings_b.transpose(0, 1)) * args.scale
+            ### Compute cross-entropy loss
+            labels = torch.tensor(range(len(scores)), dtype=torch.long, device=embeddings_a.device)  # Example a[i] should match with b[i]
+            ## One-way loss
+            loss = cross_entropy_loss(scores, labels)
+        # Backward pass
+        optimizer.zero_grad()
+        loss.backward()
+        torch.nn.utils.clip_grad_norm_(model.parameters(), max_grad_norm)
+        xm.optimizer_step(optimizer, barrier=True)
+        lr_scheduler.step()
+        #Save model
+        if (global_step+1) % args.save_steps == 0:
+            output_path = os.path.join(args.output, str(global_step+1))
+            xm.master_print("save model: "+output_path)
+            model.save_pretrained(output_path)
+    output_path = os.path.join(args.output, "final")
+    xm.master_print("save model final: "+ output_path)
+    model.save_pretrained(output_path)
+def produce_data(args, queue, filepaths, dataset_indices):
+    global_batch_size = args.batch_size*args.nprocs    #Global batch size
+    size_per_dataset = int(global_batch_size / args.datasets_per_batch)    #How many datasets per batch
+    num_same_dataset = int(size_per_dataset / args.batch_size)
+    print("producer", "global_batch_size", global_batch_size)
+    print("producer", "size_per_dataset", size_per_dataset)
+    print("producer", "num_same_dataset", num_same_dataset)
+    datasets = []
+    for filepath in filepaths:
+        if "reddit_" in filepath:       #Special dataset class for Reddit files
+            data_obj = RedditDataset(filepath)
+        else:
+            data_obj = Dataset(filepath)
+        datasets.append(iter(data_obj))
+    # Store if dataset is in a 2 col or 3 col format
+    num_cols = {idx: len(next(dataset)) for idx, dataset in enumerate(datasets)}
+    while True:
+        texts_in_batch = set()
+        batch_format = None     #2 vs 3 col format for this batch
+        #Add data from several sub datasets
+        for _ in range(args.datasets_per_batch):
+            valid_dataset = False   #Check that datasets have the same 2/3 col format
+            while not valid_dataset:
+                data_idx = random.choice(dataset_indices)
+                if batch_format is None:
+                    batch_format = num_cols[data_idx]
+                    valid_dataset = True
+                else:   #Check that this dataset has the same format
+                    valid_dataset = (batch_format == num_cols[data_idx])
+            #Get data from this dataset
+            dataset = datasets[data_idx]
+            for _ in range(num_same_dataset):
+                for _ in range(args.nprocs):
+                    batch_device = []   #A batch for one device
+                    while len(batch_device) < args.batch_size:
+                        sample = next(dataset)
+                        in_batch = False
+                        for text in sample:
+                            if text in texts_in_batch:
+                                in_batch = True
+                                break
+                        if not in_batch:
+                            for text in sample:
+                                texts_in_batch.add(text)
+                            batch_device.append(sample)
+                    queue.put(batch_device)
+class RedditDataset:
+    """
+    A class that handles the reddit data files
+    """
+    def __init__(self, filepath):
+        self.filepath = filepath
+    def __iter__(self):
+        while True:
+            with gzip.open(self.filepath, "rt") as fIn:
+                    for line in fIn:
+                        data = json.loads(line)
+                        if "response" in data and "context" in data:
+                            yield [data["response"], data["context"]]
+class Dataset:
+    """
+    A class that handles one dataset
+    """
+    def __init__(self, filepath):
+        self.filepath = filepath
+    def __iter__(self):
+        max_dataset_size = 10*1000*1000    #Cache small datasets in memory
+        dataset = []
+        data_format = None
+        while dataset is None or len(dataset) == 0:
+            with gzip.open(self.filepath, "rt") as fIn:
+                for line in fIn:
+                    data = json.loads(line)
+                    if isinstance(data, dict):
+                        data = data['texts']
+                    if data_format is None:
+                        data_format = len(data)
+                    #Ensure that all entries are of the same 2/3 col format
+                    assert len(data) == data_format
+                    if dataset is not None:
+                        dataset.append(data)
+                        if len(dataset) >= max_dataset_size:
+                            dataset = None
+                    yield data
+        # Data loaded. Now stream to the queue
+        # Shuffle for each epoch
+        while True:
+            random.shuffle(dataset)
+            for data in dataset:
+                yield data
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser()
+    parser.add_argument('--model', default='nreimers/MiniLM-L6-H384-uncased')
+    parser.add_argument('--steps', type=int, default=2000)
+    parser.add_argument('--save_steps', type=int, default=10000)
+    parser.add_argument('--batch_size', type=int, default=64)
+    parser.add_argument('--max_length', type=int, default=128)
+    parser.add_argument('--nprocs', type=int, default=8)
+    parser.add_argument('--datasets_per_batch', type=int, default=2, help="Number of datasets per batch")
+    parser.add_argument('--scale', type=float, default=20, help="Use 20 for cossim, and 1 when you work with unnormalized embeddings with dot product")
+    parser.add_argument('--data_folder', default="/data", help="Folder with your dataset files")
+    parser.add_argument('data_config', help="A data_config.json file")
+    parser.add_argument('output')
+    args = parser.parse_args()
+    # Ensure global batch size is divisble by data_sample_size
+    assert (args.batch_size*args.nprocs) % args.datasets_per_batch == 0
+    logging.info("Output: "+args.output)
+    if os.path.exists(args.output):
+        print("Output folder already exists.")
+        input("Continue?")
+    # Write train script to output path
+    os.makedirs(args.output, exist_ok=True)
+    data_config_path = os.path.join(args.output, 'data_config.json')
+    copyfile(args.data_config, data_config_path)
+    train_script_path = os.path.join(args.output, 'train_script.py')
+    copyfile(__file__, train_script_path)
+    with open(train_script_path, 'a') as fOut:
+        fOut.write("\n\n# Script was called via:\n#python " + " ".join(sys.argv))
+    #Load data config
+    with open(args.data_config) as fIn:
+        data_config = json.load(fIn)
+    queue = mp.Queue(maxsize=100*args.nprocs)
+    filepaths = []
+    dataset_indices = []
+    for idx, data in enumerate(data_config):
+        filepaths.append(os.path.join(os.path.expanduser(args.data_folder), data['name']))
+        dataset_indices.extend([idx]*data['weight'])
+    # Start producer
+    p = mp.Process(target=produce_data, args=(args, queue, filepaths, dataset_indices))
+    p.start()
+    # Run training
+    print("Start processes:", args.nprocs)
+    xmp.spawn(train_function, args=(args, queue), nprocs=args.nprocs, start_method='fork')
+    print("Training done")
+    print("It might be that not all processes exit automatically. In that case you must manually kill this process.")
+    print("With 'pkill python' you can kill all remaining python processes")
+    p.kill()
+    exit()
+# Script was called via:
+#python train_many_data_files_v2.py --steps 1000000 --batch_size 128 --model nreimers/MiniLM-L6-H384-uncased train_data_configs/all_datasets_v4.json output/all_datasets_v4_MiniLM-L6-H384-uncased-batch128

vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff