diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2568/trainer_state.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2568/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9e3c5d040995086992b3f684c39d4ee97e09ac67 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2568/trainer_state.json @@ -0,0 +1,610 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 6.0, + "eval_steps": 500, + "global_step": 2568, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8341716039180755, + "epoch": 0.11689070718877849, + "grad_norm": 1.1992322206497192, + "learning_rate": 5.366728168873013e-05, + "loss": 1.6918942260742187, + "mean_token_accuracy": 0.631798365265131, + "num_tokens": 153689.0, + "step": 50 + }, + { + "entropy": 0.9513179230690002, + "epoch": 0.23378141437755698, + "grad_norm": 1.1290454864501953, + "learning_rate": 0.00010842981402416902, + "loss": 0.8867578887939453, + "mean_token_accuracy": 0.7587438315153122, + "num_tokens": 310548.0, + "step": 100 + }, + { + "entropy": 0.8510265004634857, + "epoch": 0.3506721215663355, + "grad_norm": 0.7732136845588684, + "learning_rate": 0.00016319234635960792, + "loss": 0.7856448364257812, + "mean_token_accuracy": 0.7769897204637527, + "num_tokens": 468560.0, + "step": 150 + }, + { + "entropy": 0.8142367601394653, + "epoch": 0.46756282875511396, + "grad_norm": 0.7580806016921997, + "learning_rate": 0.00021795487869504684, + "loss": 0.7461666870117187, + "mean_token_accuracy": 0.7888966089487076, + "num_tokens": 619552.0, + "step": 200 + }, + { + "entropy": 0.772578022480011, + "epoch": 0.5844535359438925, + "grad_norm": 0.5330358147621155, + "learning_rate": 0.00027271741103048575, + "loss": 0.7142900848388671, + "mean_token_accuracy": 0.7959607627987861, + "num_tokens": 771972.0, + "step": 250 + }, + { + "entropy": 0.7481484657526016, + "epoch": 0.701344243132671, + "grad_norm": 0.8242517709732056, + "learning_rate": 0.00032747994336592464, + "loss": 0.6911422729492187, + "mean_token_accuracy": 0.8028205358982086, + "num_tokens": 928088.0, + "step": 300 + }, + { + "entropy": 0.741372903585434, + "epoch": 0.8182349503214494, + "grad_norm": 0.7722981572151184, + "learning_rate": 0.00038224247570136353, + "loss": 0.6908904266357422, + "mean_token_accuracy": 0.8018302822113037, + "num_tokens": 1086836.0, + "step": 350 + }, + { + "entropy": 0.7257227802276611, + "epoch": 0.9351256575102279, + "grad_norm": 0.7969573140144348, + "learning_rate": 0.0004370050080368024, + "loss": 0.6714310455322265, + "mean_token_accuracy": 0.8057891410589219, + "num_tokens": 1241227.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8317563653766334, + "eval_loss": 0.8135058283805847, + "eval_mean_token_accuracy": 0.7755894215850087, + "eval_num_tokens": 1325643.0, + "eval_runtime": 37.4951, + "eval_samples_per_second": 32.831, + "eval_steps_per_second": 4.107, + "step": 428 + }, + { + "entropy": 0.7179872094087265, + "epoch": 1.0514319111630626, + "grad_norm": 0.6224768161773682, + "learning_rate": 0.00046873290101955413, + "loss": 0.6683222961425781, + "mean_token_accuracy": 0.8054787133207273, + "num_tokens": 1393922.0, + "step": 450 + }, + { + "entropy": 0.6991294291615486, + "epoch": 1.1683226183518411, + "grad_norm": 0.6813111901283264, + "learning_rate": 0.00046837443306086947, + "loss": 0.6493977355957031, + "mean_token_accuracy": 0.8089349576830864, + "num_tokens": 1551442.0, + "step": 500 + }, + { + "entropy": 0.6951853144168854, + "epoch": 1.2852133255406195, + "grad_norm": 0.617091953754425, + "learning_rate": 0.0004676269147738558, + "loss": 0.6484123229980469, + "mean_token_accuracy": 0.8088837671279907, + "num_tokens": 1706697.0, + "step": 550 + }, + { + "entropy": 0.6852494943141937, + "epoch": 1.4021040327293979, + "grad_norm": 0.4677598774433136, + "learning_rate": 0.0004664915890374708, + "loss": 0.6416233062744141, + "mean_token_accuracy": 0.8106312158703805, + "num_tokens": 1865246.0, + "step": 600 + }, + { + "entropy": 0.6703594943881035, + "epoch": 1.5189947399181765, + "grad_norm": 0.5708025693893433, + "learning_rate": 0.0004649703435278991, + "loss": 0.6273183441162109, + "mean_token_accuracy": 0.8155005398392677, + "num_tokens": 2022060.0, + "step": 650 + }, + { + "entropy": 0.6676286320388317, + "epoch": 1.635885447106955, + "grad_norm": 0.5555347800254822, + "learning_rate": 0.00046306570757996264, + "loss": 0.6276700973510743, + "mean_token_accuracy": 0.8154828292131424, + "num_tokens": 2178909.0, + "step": 700 + }, + { + "entropy": 0.6669242936372757, + "epoch": 1.7527761542957334, + "grad_norm": 0.5260112285614014, + "learning_rate": 0.0004607808479816624, + "loss": 0.6228141784667969, + "mean_token_accuracy": 0.8164840793609619, + "num_tokens": 2329314.0, + "step": 750 + }, + { + "entropy": 0.6598560312390327, + "epoch": 1.869666861484512, + "grad_norm": 0.5677736401557922, + "learning_rate": 0.0004581195637088436, + "loss": 0.6214292907714843, + "mean_token_accuracy": 0.8171795177459716, + "num_tokens": 2478559.0, + "step": 800 + }, + { + "entropy": 0.6687870016694069, + "epoch": 1.9865575686732906, + "grad_norm": 0.6449615955352783, + "learning_rate": 0.00045508627960873823, + "loss": 0.6243909454345703, + "mean_token_accuracy": 0.8147780740261078, + "num_tokens": 2631937.0, + "step": 850 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7786144471013701, + "eval_loss": 0.7599140405654907, + "eval_mean_token_accuracy": 0.7935443106409791, + "eval_num_tokens": 2651286.0, + "eval_runtime": 39.0048, + "eval_samples_per_second": 31.56, + "eval_steps_per_second": 3.948, + "step": 856 + }, + { + "entropy": 0.5821921329701966, + "epoch": 2.102863822326125, + "grad_norm": 0.6116960048675537, + "learning_rate": 0.00045168603904288863, + "loss": 0.5406004714965821, + "mean_token_accuracy": 0.8334921918921734, + "num_tokens": 2795200.0, + "step": 900 + }, + { + "entropy": 0.5937331764400006, + "epoch": 2.2197545295149035, + "grad_norm": 0.5440122485160828, + "learning_rate": 0.00044792449550168286, + "loss": 0.5515737533569336, + "mean_token_accuracy": 0.8314691257476806, + "num_tokens": 2950534.0, + "step": 950 + }, + { + "entropy": 0.5877273553609847, + "epoch": 2.3366452367036823, + "grad_norm": 0.6505260467529297, + "learning_rate": 0.0004438079032044453, + "loss": 0.5507744979858399, + "mean_token_accuracy": 0.8314063146710395, + "num_tokens": 3101471.0, + "step": 1000 + }, + { + "entropy": 0.5897100016474723, + "epoch": 2.4535359438924607, + "grad_norm": 0.54567551612854, + "learning_rate": 0.0004393431067007111, + "loss": 0.5528768157958984, + "mean_token_accuracy": 0.8315245220065117, + "num_tokens": 3258554.0, + "step": 1050 + }, + { + "entropy": 0.5831590622663498, + "epoch": 2.570426651081239, + "grad_norm": 0.44826796650886536, + "learning_rate": 0.00043453752948997376, + "loss": 0.5442767333984375, + "mean_token_accuracy": 0.8327079233527184, + "num_tokens": 3416088.0, + "step": 1100 + }, + { + "entropy": 0.5839906217157841, + "epoch": 2.6873173582700174, + "grad_norm": 0.5067696571350098, + "learning_rate": 0.0004293991616788285, + "loss": 0.5485663604736328, + "mean_token_accuracy": 0.8324281191825866, + "num_tokens": 3567251.0, + "step": 1150 + }, + { + "entropy": 0.5870274990797043, + "epoch": 2.8042080654587958, + "grad_norm": 0.44032806158065796, + "learning_rate": 0.00042393654669603217, + "loss": 0.5524474716186524, + "mean_token_accuracy": 0.8313613015413285, + "num_tokens": 3719464.0, + "step": 1200 + }, + { + "entropy": 0.5864466108381748, + "epoch": 2.9210987726475746, + "grad_norm": 0.5520864129066467, + "learning_rate": 0.00041815876708756964, + "loss": 0.5490786361694336, + "mean_token_accuracy": 0.8314764249324799, + "num_tokens": 3874423.0, + "step": 1250 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6819627613990338, + "eval_loss": 0.7593190670013428, + "eval_mean_token_accuracy": 0.7857603395914102, + "eval_num_tokens": 3976929.0, + "eval_runtime": 39.3654, + "eval_samples_per_second": 31.271, + "eval_steps_per_second": 3.912, + "step": 1284 + }, + { + "entropy": 0.5587641523411525, + "epoch": 3.037405026300409, + "grad_norm": 0.622401773929596, + "learning_rate": 0.0004120754294153441, + "loss": 0.518932762145996, + "mean_token_accuracy": 0.8399260457436643, + "num_tokens": 4027609.0, + "step": 1300 + }, + { + "entropy": 0.5028628017008304, + "epoch": 3.1542957334891875, + "grad_norm": 0.577396810054779, + "learning_rate": 0.00040569664828459917, + "loss": 0.4589382171630859, + "mean_token_accuracy": 0.8538418188691139, + "num_tokens": 4180560.0, + "step": 1350 + }, + { + "entropy": 0.5227407096326351, + "epoch": 3.2711864406779663, + "grad_norm": 0.5113071203231812, + "learning_rate": 0.00039903302952663176, + "loss": 0.47863777160644533, + "mean_token_accuracy": 0.847364938557148, + "num_tokens": 4329954.0, + "step": 1400 + }, + { + "entropy": 0.5050077450275421, + "epoch": 3.3880771478667446, + "grad_norm": 0.48977184295654297, + "learning_rate": 0.0003920956525647558, + "loss": 0.46841480255126955, + "mean_token_accuracy": 0.8519425508379936, + "num_tokens": 4484354.0, + "step": 1450 + }, + { + "entropy": 0.5163291451334954, + "epoch": 3.504967855055523, + "grad_norm": 0.509864330291748, + "learning_rate": 0.000384896051992837, + "loss": 0.47587432861328127, + "mean_token_accuracy": 0.8486990982294083, + "num_tokens": 4638919.0, + "step": 1500 + }, + { + "entropy": 0.5134807989001274, + "epoch": 3.6218585622443014, + "grad_norm": 0.46544620394706726, + "learning_rate": 0.00037744619839702735, + "loss": 0.47692710876464844, + "mean_token_accuracy": 0.8489899519085884, + "num_tokens": 4795540.0, + "step": 1550 + }, + { + "entropy": 0.5216251534223556, + "epoch": 3.73874926943308, + "grad_norm": 0.5167004466056824, + "learning_rate": 0.0003697584784525874, + "loss": 0.4837848663330078, + "mean_token_accuracy": 0.847066233754158, + "num_tokens": 4951177.0, + "step": 1600 + }, + { + "entropy": 0.5254101701080799, + "epoch": 3.8556399766218585, + "grad_norm": 0.5504006743431091, + "learning_rate": 0.00036184567432888745, + "loss": 0.48759506225585936, + "mean_token_accuracy": 0.8457375919818878, + "num_tokens": 5104814.0, + "step": 1650 + }, + { + "entropy": 0.5107753933966159, + "epoch": 3.972530683810637, + "grad_norm": 0.421975314617157, + "learning_rate": 0.0003537209424368311, + "loss": 0.4743759536743164, + "mean_token_accuracy": 0.8505110186338425, + "num_tokens": 5265185.0, + "step": 1700 + }, + { + "epoch": 4.0, + "eval_entropy": 0.665543915002377, + "eval_loss": 0.7487082481384277, + "eval_mean_token_accuracy": 0.7999678904359991, + "eval_num_tokens": 5302572.0, + "eval_runtime": 38.5439, + "eval_samples_per_second": 31.938, + "eval_steps_per_second": 3.995, + "step": 1712 + }, + { + "entropy": 0.4414465431891494, + "epoch": 4.088836937463472, + "grad_norm": 0.5308493971824646, + "learning_rate": 0.0003453977915540383, + "loss": 0.39650299072265627, + "mean_token_accuracy": 0.8703725772287378, + "num_tokens": 5419002.0, + "step": 1750 + }, + { + "entropy": 0.42591195791959763, + "epoch": 4.20572764465225, + "grad_norm": 0.6111563444137573, + "learning_rate": 0.00033689006036415585, + "loss": 0.37838283538818357, + "mean_token_accuracy": 0.874235480427742, + "num_tokens": 5572969.0, + "step": 1800 + }, + { + "entropy": 0.44086351931095125, + "epoch": 4.322618351841029, + "grad_norm": 0.5524880290031433, + "learning_rate": 0.0003282118944476435, + "loss": 0.3920352554321289, + "mean_token_accuracy": 0.8694414687156677, + "num_tokens": 5733805.0, + "step": 1850 + }, + { + "entropy": 0.4380896310508251, + "epoch": 4.439509059029807, + "grad_norm": 0.5023863315582275, + "learning_rate": 0.0003193777227622898, + "loss": 0.3916081237792969, + "mean_token_accuracy": 0.8697139009833336, + "num_tokens": 5888694.0, + "step": 1900 + }, + { + "entropy": 0.44005885019898416, + "epoch": 4.556399766218585, + "grad_norm": 0.455997109413147, + "learning_rate": 0.000310402233652564, + "loss": 0.3955466842651367, + "mean_token_accuracy": 0.8693471103906631, + "num_tokens": 6042655.0, + "step": 1950 + }, + { + "entropy": 0.444269048422575, + "epoch": 4.673290473407365, + "grad_norm": 0.4542011320590973, + "learning_rate": 0.00030130035042769316, + "loss": 0.3988466262817383, + "mean_token_accuracy": 0.8680117425322532, + "num_tokens": 6203980.0, + "step": 2000 + }, + { + "entropy": 0.43520899042487143, + "epoch": 4.790181180596143, + "grad_norm": 0.49051445722579956, + "learning_rate": 0.0002920872065490688, + "loss": 0.39819797515869143, + "mean_token_accuracy": 0.8678940117359162, + "num_tokens": 6354604.0, + "step": 2050 + }, + { + "entropy": 0.429456724524498, + "epoch": 4.907071887784921, + "grad_norm": 0.5267980098724365, + "learning_rate": 0.0002827781204682396, + "loss": 0.39413917541503907, + "mean_token_accuracy": 0.870051506459713, + "num_tokens": 6507484.0, + "step": 2100 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5826305385146823, + "eval_loss": 0.7694042921066284, + "eval_mean_token_accuracy": 0.7950637824349589, + "eval_num_tokens": 6628215.0, + "eval_runtime": 38.5641, + "eval_samples_per_second": 31.921, + "eval_steps_per_second": 3.993, + "step": 2140 + }, + { + "entropy": 0.41423329688496324, + "epoch": 5.023378141437756, + "grad_norm": 0.6393507719039917, + "learning_rate": 0.0002733885701573256, + "loss": 0.3733938598632813, + "mean_token_accuracy": 0.8762160557598325, + "num_tokens": 6662015.0, + "step": 2150 + }, + { + "entropy": 0.32116023637354374, + "epoch": 5.140268848626534, + "grad_norm": 0.5128306150436401, + "learning_rate": 0.00026393416737420213, + "loss": 0.2751211166381836, + "mean_token_accuracy": 0.9052674892544746, + "num_tokens": 6817989.0, + "step": 2200 + }, + { + "entropy": 0.33186347484588624, + "epoch": 5.257159555815313, + "grad_norm": 0.6007080078125, + "learning_rate": 0.0002544306317052408, + "loss": 0.2850212097167969, + "mean_token_accuracy": 0.9016379952430725, + "num_tokens": 6977021.0, + "step": 2250 + }, + { + "entropy": 0.3474555689096451, + "epoch": 5.374050263004091, + "grad_norm": 0.6231185793876648, + "learning_rate": 0.0002448937644287679, + "loss": 0.29555959701538087, + "mean_token_accuracy": 0.8966628012061119, + "num_tokens": 7133360.0, + "step": 2300 + }, + { + "entropy": 0.346554354429245, + "epoch": 5.490940970192869, + "grad_norm": 0.5146846175193787, + "learning_rate": 0.0002353394222426952, + "loss": 0.29524572372436525, + "mean_token_accuracy": 0.8978548383712769, + "num_tokens": 7282372.0, + "step": 2350 + }, + { + "entropy": 0.3467242659628391, + "epoch": 5.607831677381649, + "grad_norm": 0.6118131875991821, + "learning_rate": 0.00022578349090000624, + "loss": 0.2970552635192871, + "mean_token_accuracy": 0.8985717830061912, + "num_tokens": 7435766.0, + "step": 2400 + }, + { + "entropy": 0.34968974225223065, + "epoch": 5.724722384570427, + "grad_norm": 0.6346496939659119, + "learning_rate": 0.0002162418587959333, + "loss": 0.29940626144409177, + "mean_token_accuracy": 0.8972864350676537, + "num_tokens": 7578408.0, + "step": 2450 + }, + { + "entropy": 0.35084108062088487, + "epoch": 5.841613091759205, + "grad_norm": 0.6566210389137268, + "learning_rate": 0.00020673039055074148, + "loss": 0.302426700592041, + "mean_token_accuracy": 0.8953999072313309, + "num_tokens": 7734599.0, + "step": 2500 + }, + { + "entropy": 0.33895430400967597, + "epoch": 5.958503798947984, + "grad_norm": 0.517574667930603, + "learning_rate": 0.00019726490063204247, + "loss": 0.292838020324707, + "mean_token_accuracy": 0.8984868070483207, + "num_tokens": 7898028.0, + "step": 2550 + }, + { + "epoch": 6.0, + "eval_entropy": 0.5211211553254684, + "eval_loss": 0.8269560933113098, + "eval_mean_token_accuracy": 0.7899065927251593, + "eval_num_tokens": 7953858.0, + "eval_runtime": 39.369, + "eval_samples_per_second": 31.268, + "eval_steps_per_second": 3.912, + "step": 2568 + } + ], + "logging_steps": 50, + "max_steps": 4280, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.1913679730121984e+17, + "train_batch_size": 4, + "trial_name": null, + "trial_params": null +} diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/README.md b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/adapter_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..0c2a7efe7af0ceadd86d0f851dc87d465191c777 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.04180832058159435, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "o_proj", + "q_proj", + "gate_proj", + "k_proj", + "down_proj", + "up_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/chat_template.jinja b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/tokenizer_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/trainer_state.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d1c6e15c5e11dd2da4259f069c11ea4d81fd4156 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-2996/trainer_state.json @@ -0,0 +1,701 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 7.0, + "eval_steps": 500, + "global_step": 2996, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8341716039180755, + "epoch": 0.11689070718877849, + "grad_norm": 1.1992322206497192, + "learning_rate": 5.366728168873013e-05, + "loss": 1.6918942260742187, + "mean_token_accuracy": 0.631798365265131, + "num_tokens": 153689.0, + "step": 50 + }, + { + "entropy": 0.9513179230690002, + "epoch": 0.23378141437755698, + "grad_norm": 1.1290454864501953, + "learning_rate": 0.00010842981402416902, + "loss": 0.8867578887939453, + "mean_token_accuracy": 0.7587438315153122, + "num_tokens": 310548.0, + "step": 100 + }, + { + "entropy": 0.8510265004634857, + "epoch": 0.3506721215663355, + "grad_norm": 0.7732136845588684, + "learning_rate": 0.00016319234635960792, + "loss": 0.7856448364257812, + "mean_token_accuracy": 0.7769897204637527, + "num_tokens": 468560.0, + "step": 150 + }, + { + "entropy": 0.8142367601394653, + "epoch": 0.46756282875511396, + "grad_norm": 0.7580806016921997, + "learning_rate": 0.00021795487869504684, + "loss": 0.7461666870117187, + "mean_token_accuracy": 0.7888966089487076, + "num_tokens": 619552.0, + "step": 200 + }, + { + "entropy": 0.772578022480011, + "epoch": 0.5844535359438925, + "grad_norm": 0.5330358147621155, + "learning_rate": 0.00027271741103048575, + "loss": 0.7142900848388671, + "mean_token_accuracy": 0.7959607627987861, + "num_tokens": 771972.0, + "step": 250 + }, + { + "entropy": 0.7481484657526016, + "epoch": 0.701344243132671, + "grad_norm": 0.8242517709732056, + "learning_rate": 0.00032747994336592464, + "loss": 0.6911422729492187, + "mean_token_accuracy": 0.8028205358982086, + "num_tokens": 928088.0, + "step": 300 + }, + { + "entropy": 0.741372903585434, + "epoch": 0.8182349503214494, + "grad_norm": 0.7722981572151184, + "learning_rate": 0.00038224247570136353, + "loss": 0.6908904266357422, + "mean_token_accuracy": 0.8018302822113037, + "num_tokens": 1086836.0, + "step": 350 + }, + { + "entropy": 0.7257227802276611, + "epoch": 0.9351256575102279, + "grad_norm": 0.7969573140144348, + "learning_rate": 0.0004370050080368024, + "loss": 0.6714310455322265, + "mean_token_accuracy": 0.8057891410589219, + "num_tokens": 1241227.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8317563653766334, + "eval_loss": 0.8135058283805847, + "eval_mean_token_accuracy": 0.7755894215850087, + "eval_num_tokens": 1325643.0, + "eval_runtime": 37.4951, + "eval_samples_per_second": 32.831, + "eval_steps_per_second": 4.107, + "step": 428 + }, + { + "entropy": 0.7179872094087265, + "epoch": 1.0514319111630626, + "grad_norm": 0.6224768161773682, + "learning_rate": 0.00046873290101955413, + "loss": 0.6683222961425781, + "mean_token_accuracy": 0.8054787133207273, + "num_tokens": 1393922.0, + "step": 450 + }, + { + "entropy": 0.6991294291615486, + "epoch": 1.1683226183518411, + "grad_norm": 0.6813111901283264, + "learning_rate": 0.00046837443306086947, + "loss": 0.6493977355957031, + "mean_token_accuracy": 0.8089349576830864, + "num_tokens": 1551442.0, + "step": 500 + }, + { + "entropy": 0.6951853144168854, + "epoch": 1.2852133255406195, + "grad_norm": 0.617091953754425, + "learning_rate": 0.0004676269147738558, + "loss": 0.6484123229980469, + "mean_token_accuracy": 0.8088837671279907, + "num_tokens": 1706697.0, + "step": 550 + }, + { + "entropy": 0.6852494943141937, + "epoch": 1.4021040327293979, + "grad_norm": 0.4677598774433136, + "learning_rate": 0.0004664915890374708, + "loss": 0.6416233062744141, + "mean_token_accuracy": 0.8106312158703805, + "num_tokens": 1865246.0, + "step": 600 + }, + { + "entropy": 0.6703594943881035, + "epoch": 1.5189947399181765, + "grad_norm": 0.5708025693893433, + "learning_rate": 0.0004649703435278991, + "loss": 0.6273183441162109, + "mean_token_accuracy": 0.8155005398392677, + "num_tokens": 2022060.0, + "step": 650 + }, + { + "entropy": 0.6676286320388317, + "epoch": 1.635885447106955, + "grad_norm": 0.5555347800254822, + "learning_rate": 0.00046306570757996264, + "loss": 0.6276700973510743, + "mean_token_accuracy": 0.8154828292131424, + "num_tokens": 2178909.0, + "step": 700 + }, + { + "entropy": 0.6669242936372757, + "epoch": 1.7527761542957334, + "grad_norm": 0.5260112285614014, + "learning_rate": 0.0004607808479816624, + "loss": 0.6228141784667969, + "mean_token_accuracy": 0.8164840793609619, + "num_tokens": 2329314.0, + "step": 750 + }, + { + "entropy": 0.6598560312390327, + "epoch": 1.869666861484512, + "grad_norm": 0.5677736401557922, + "learning_rate": 0.0004581195637088436, + "loss": 0.6214292907714843, + "mean_token_accuracy": 0.8171795177459716, + "num_tokens": 2478559.0, + "step": 800 + }, + { + "entropy": 0.6687870016694069, + "epoch": 1.9865575686732906, + "grad_norm": 0.6449615955352783, + "learning_rate": 0.00045508627960873823, + "loss": 0.6243909454345703, + "mean_token_accuracy": 0.8147780740261078, + "num_tokens": 2631937.0, + "step": 850 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7786144471013701, + "eval_loss": 0.7599140405654907, + "eval_mean_token_accuracy": 0.7935443106409791, + "eval_num_tokens": 2651286.0, + "eval_runtime": 39.0048, + "eval_samples_per_second": 31.56, + "eval_steps_per_second": 3.948, + "step": 856 + }, + { + "entropy": 0.5821921329701966, + "epoch": 2.102863822326125, + "grad_norm": 0.6116960048675537, + "learning_rate": 0.00045168603904288863, + "loss": 0.5406004714965821, + "mean_token_accuracy": 0.8334921918921734, + "num_tokens": 2795200.0, + "step": 900 + }, + { + "entropy": 0.5937331764400006, + "epoch": 2.2197545295149035, + "grad_norm": 0.5440122485160828, + "learning_rate": 0.00044792449550168286, + "loss": 0.5515737533569336, + "mean_token_accuracy": 0.8314691257476806, + "num_tokens": 2950534.0, + "step": 950 + }, + { + "entropy": 0.5877273553609847, + "epoch": 2.3366452367036823, + "grad_norm": 0.6505260467529297, + "learning_rate": 0.0004438079032044453, + "loss": 0.5507744979858399, + "mean_token_accuracy": 0.8314063146710395, + "num_tokens": 3101471.0, + "step": 1000 + }, + { + "entropy": 0.5897100016474723, + "epoch": 2.4535359438924607, + "grad_norm": 0.54567551612854, + "learning_rate": 0.0004393431067007111, + "loss": 0.5528768157958984, + "mean_token_accuracy": 0.8315245220065117, + "num_tokens": 3258554.0, + "step": 1050 + }, + { + "entropy": 0.5831590622663498, + "epoch": 2.570426651081239, + "grad_norm": 0.44826796650886536, + "learning_rate": 0.00043453752948997376, + "loss": 0.5442767333984375, + "mean_token_accuracy": 0.8327079233527184, + "num_tokens": 3416088.0, + "step": 1100 + }, + { + "entropy": 0.5839906217157841, + "epoch": 2.6873173582700174, + "grad_norm": 0.5067696571350098, + "learning_rate": 0.0004293991616788285, + "loss": 0.5485663604736328, + "mean_token_accuracy": 0.8324281191825866, + "num_tokens": 3567251.0, + "step": 1150 + }, + { + "entropy": 0.5870274990797043, + "epoch": 2.8042080654587958, + "grad_norm": 0.44032806158065796, + "learning_rate": 0.00042393654669603217, + "loss": 0.5524474716186524, + "mean_token_accuracy": 0.8313613015413285, + "num_tokens": 3719464.0, + "step": 1200 + }, + { + "entropy": 0.5864466108381748, + "epoch": 2.9210987726475746, + "grad_norm": 0.5520864129066467, + "learning_rate": 0.00041815876708756964, + "loss": 0.5490786361694336, + "mean_token_accuracy": 0.8314764249324799, + "num_tokens": 3874423.0, + "step": 1250 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6819627613990338, + "eval_loss": 0.7593190670013428, + "eval_mean_token_accuracy": 0.7857603395914102, + "eval_num_tokens": 3976929.0, + "eval_runtime": 39.3654, + "eval_samples_per_second": 31.271, + "eval_steps_per_second": 3.912, + "step": 1284 + }, + { + "entropy": 0.5587641523411525, + "epoch": 3.037405026300409, + "grad_norm": 0.622401773929596, + "learning_rate": 0.0004120754294153441, + "loss": 0.518932762145996, + "mean_token_accuracy": 0.8399260457436643, + "num_tokens": 4027609.0, + "step": 1300 + }, + { + "entropy": 0.5028628017008304, + "epoch": 3.1542957334891875, + "grad_norm": 0.577396810054779, + "learning_rate": 0.00040569664828459917, + "loss": 0.4589382171630859, + "mean_token_accuracy": 0.8538418188691139, + "num_tokens": 4180560.0, + "step": 1350 + }, + { + "entropy": 0.5227407096326351, + "epoch": 3.2711864406779663, + "grad_norm": 0.5113071203231812, + "learning_rate": 0.00039903302952663176, + "loss": 0.47863777160644533, + "mean_token_accuracy": 0.847364938557148, + "num_tokens": 4329954.0, + "step": 1400 + }, + { + "entropy": 0.5050077450275421, + "epoch": 3.3880771478667446, + "grad_norm": 0.48977184295654297, + "learning_rate": 0.0003920956525647558, + "loss": 0.46841480255126955, + "mean_token_accuracy": 0.8519425508379936, + "num_tokens": 4484354.0, + "step": 1450 + }, + { + "entropy": 0.5163291451334954, + "epoch": 3.504967855055523, + "grad_norm": 0.509864330291748, + "learning_rate": 0.000384896051992837, + "loss": 0.47587432861328127, + "mean_token_accuracy": 0.8486990982294083, + "num_tokens": 4638919.0, + "step": 1500 + }, + { + "entropy": 0.5134807989001274, + "epoch": 3.6218585622443014, + "grad_norm": 0.46544620394706726, + "learning_rate": 0.00037744619839702735, + "loss": 0.47692710876464844, + "mean_token_accuracy": 0.8489899519085884, + "num_tokens": 4795540.0, + "step": 1550 + }, + { + "entropy": 0.5216251534223556, + "epoch": 3.73874926943308, + "grad_norm": 0.5167004466056824, + "learning_rate": 0.0003697584784525874, + "loss": 0.4837848663330078, + "mean_token_accuracy": 0.847066233754158, + "num_tokens": 4951177.0, + "step": 1600 + }, + { + "entropy": 0.5254101701080799, + "epoch": 3.8556399766218585, + "grad_norm": 0.5504006743431091, + "learning_rate": 0.00036184567432888745, + "loss": 0.48759506225585936, + "mean_token_accuracy": 0.8457375919818878, + "num_tokens": 5104814.0, + "step": 1650 + }, + { + "entropy": 0.5107753933966159, + "epoch": 3.972530683810637, + "grad_norm": 0.421975314617157, + "learning_rate": 0.0003537209424368311, + "loss": 0.4743759536743164, + "mean_token_accuracy": 0.8505110186338425, + "num_tokens": 5265185.0, + "step": 1700 + }, + { + "epoch": 4.0, + "eval_entropy": 0.665543915002377, + "eval_loss": 0.7487082481384277, + "eval_mean_token_accuracy": 0.7999678904359991, + "eval_num_tokens": 5302572.0, + "eval_runtime": 38.5439, + "eval_samples_per_second": 31.938, + "eval_steps_per_second": 3.995, + "step": 1712 + }, + { + "entropy": 0.4414465431891494, + "epoch": 4.088836937463472, + "grad_norm": 0.5308493971824646, + "learning_rate": 0.0003453977915540383, + "loss": 0.39650299072265627, + "mean_token_accuracy": 0.8703725772287378, + "num_tokens": 5419002.0, + "step": 1750 + }, + { + "entropy": 0.42591195791959763, + "epoch": 4.20572764465225, + "grad_norm": 0.6111563444137573, + "learning_rate": 0.00033689006036415585, + "loss": 0.37838283538818357, + "mean_token_accuracy": 0.874235480427742, + "num_tokens": 5572969.0, + "step": 1800 + }, + { + "entropy": 0.44086351931095125, + "epoch": 4.322618351841029, + "grad_norm": 0.5524880290031433, + "learning_rate": 0.0003282118944476435, + "loss": 0.3920352554321289, + "mean_token_accuracy": 0.8694414687156677, + "num_tokens": 5733805.0, + "step": 1850 + }, + { + "entropy": 0.4380896310508251, + "epoch": 4.439509059029807, + "grad_norm": 0.5023863315582275, + "learning_rate": 0.0003193777227622898, + "loss": 0.3916081237792969, + "mean_token_accuracy": 0.8697139009833336, + "num_tokens": 5888694.0, + "step": 1900 + }, + { + "entropy": 0.44005885019898416, + "epoch": 4.556399766218585, + "grad_norm": 0.455997109413147, + "learning_rate": 0.000310402233652564, + "loss": 0.3955466842651367, + "mean_token_accuracy": 0.8693471103906631, + "num_tokens": 6042655.0, + "step": 1950 + }, + { + "entropy": 0.444269048422575, + "epoch": 4.673290473407365, + "grad_norm": 0.4542011320590973, + "learning_rate": 0.00030130035042769316, + "loss": 0.3988466262817383, + "mean_token_accuracy": 0.8680117425322532, + "num_tokens": 6203980.0, + "step": 2000 + }, + { + "entropy": 0.43520899042487143, + "epoch": 4.790181180596143, + "grad_norm": 0.49051445722579956, + "learning_rate": 0.0002920872065490688, + "loss": 0.39819797515869143, + "mean_token_accuracy": 0.8678940117359162, + "num_tokens": 6354604.0, + "step": 2050 + }, + { + "entropy": 0.429456724524498, + "epoch": 4.907071887784921, + "grad_norm": 0.5267980098724365, + "learning_rate": 0.0002827781204682396, + "loss": 0.39413917541503907, + "mean_token_accuracy": 0.870051506459713, + "num_tokens": 6507484.0, + "step": 2100 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5826305385146823, + "eval_loss": 0.7694042921066284, + "eval_mean_token_accuracy": 0.7950637824349589, + "eval_num_tokens": 6628215.0, + "eval_runtime": 38.5641, + "eval_samples_per_second": 31.921, + "eval_steps_per_second": 3.993, + "step": 2140 + }, + { + "entropy": 0.41423329688496324, + "epoch": 5.023378141437756, + "grad_norm": 0.6393507719039917, + "learning_rate": 0.0002733885701573256, + "loss": 0.3733938598632813, + "mean_token_accuracy": 0.8762160557598325, + "num_tokens": 6662015.0, + "step": 2150 + }, + { + "entropy": 0.32116023637354374, + "epoch": 5.140268848626534, + "grad_norm": 0.5128306150436401, + "learning_rate": 0.00026393416737420213, + "loss": 0.2751211166381836, + "mean_token_accuracy": 0.9052674892544746, + "num_tokens": 6817989.0, + "step": 2200 + }, + { + "entropy": 0.33186347484588624, + "epoch": 5.257159555815313, + "grad_norm": 0.6007080078125, + "learning_rate": 0.0002544306317052408, + "loss": 0.2850212097167969, + "mean_token_accuracy": 0.9016379952430725, + "num_tokens": 6977021.0, + "step": 2250 + }, + { + "entropy": 0.3474555689096451, + "epoch": 5.374050263004091, + "grad_norm": 0.6231185793876648, + "learning_rate": 0.0002448937644287679, + "loss": 0.29555959701538087, + "mean_token_accuracy": 0.8966628012061119, + "num_tokens": 7133360.0, + "step": 2300 + }, + { + "entropy": 0.346554354429245, + "epoch": 5.490940970192869, + "grad_norm": 0.5146846175193787, + "learning_rate": 0.0002353394222426952, + "loss": 0.29524572372436525, + "mean_token_accuracy": 0.8978548383712769, + "num_tokens": 7282372.0, + "step": 2350 + }, + { + "entropy": 0.3467242659628391, + "epoch": 5.607831677381649, + "grad_norm": 0.6118131875991821, + "learning_rate": 0.00022578349090000624, + "loss": 0.2970552635192871, + "mean_token_accuracy": 0.8985717830061912, + "num_tokens": 7435766.0, + "step": 2400 + }, + { + "entropy": 0.34968974225223065, + "epoch": 5.724722384570427, + "grad_norm": 0.6346496939659119, + "learning_rate": 0.0002162418587959333, + "loss": 0.29940626144409177, + "mean_token_accuracy": 0.8972864350676537, + "num_tokens": 7578408.0, + "step": 2450 + }, + { + "entropy": 0.35084108062088487, + "epoch": 5.841613091759205, + "grad_norm": 0.6566210389137268, + "learning_rate": 0.00020673039055074148, + "loss": 0.302426700592041, + "mean_token_accuracy": 0.8953999072313309, + "num_tokens": 7734599.0, + "step": 2500 + }, + { + "entropy": 0.33895430400967597, + "epoch": 5.958503798947984, + "grad_norm": 0.517574667930603, + "learning_rate": 0.00019726490063204247, + "loss": 0.292838020324707, + "mean_token_accuracy": 0.8984868070483207, + "num_tokens": 7898028.0, + "step": 2550 + }, + { + "epoch": 6.0, + "eval_entropy": 0.5211211553254684, + "eval_loss": 0.8269560933113098, + "eval_mean_token_accuracy": 0.7899065927251593, + "eval_num_tokens": 7953858.0, + "eval_runtime": 39.369, + "eval_samples_per_second": 31.268, + "eval_steps_per_second": 3.912, + "step": 2568 + }, + { + "entropy": 0.2763742347009218, + "epoch": 6.074810052600818, + "grad_norm": 0.5116816163063049, + "learning_rate": 0.00018786112706049623, + "loss": 0.22524795532226563, + "mean_token_accuracy": 0.9218554844209297, + "num_tokens": 8052646.0, + "step": 2600 + }, + { + "entropy": 0.23341606348752975, + "epoch": 6.1917007597895966, + "grad_norm": 0.6335962414741516, + "learning_rate": 0.00017853470524261842, + "loss": 0.18153846740722657, + "mean_token_accuracy": 0.935965863764286, + "num_tokens": 8212323.0, + "step": 2650 + }, + { + "entropy": 0.23515729174017908, + "epoch": 6.308591466978375, + "grad_norm": 0.6312636733055115, + "learning_rate": 0.0001693011419742025, + "loss": 0.18562854766845704, + "mean_token_accuracy": 0.9342144966125489, + "num_tokens": 8366715.0, + "step": 2700 + }, + { + "entropy": 0.23566059060394765, + "epoch": 6.425482174167154, + "grad_norm": 0.5869485139846802, + "learning_rate": 0.0001601757896575784, + "loss": 0.1867989158630371, + "mean_token_accuracy": 0.934665755033493, + "num_tokens": 8515955.0, + "step": 2750 + }, + { + "entropy": 0.2446490554511547, + "epoch": 6.5423728813559325, + "grad_norm": 0.5957017540931702, + "learning_rate": 0.00015117382077557817, + "loss": 0.19312274932861329, + "mean_token_accuracy": 0.9308532625436783, + "num_tokens": 8672361.0, + "step": 2800 + }, + { + "entropy": 0.24066772796213626, + "epoch": 6.659263588544711, + "grad_norm": 0.6069761514663696, + "learning_rate": 0.0001423102026646483, + "loss": 0.19540096282958985, + "mean_token_accuracy": 0.9300534284114838, + "num_tokens": 8827819.0, + "step": 2850 + }, + { + "entropy": 0.2378876845538616, + "epoch": 6.776154295733489, + "grad_norm": 0.6069548726081848, + "learning_rate": 0.00013359967262905405, + "loss": 0.1942250633239746, + "mean_token_accuracy": 0.9310428243875504, + "num_tokens": 8979535.0, + "step": 2900 + }, + { + "entropy": 0.23810458809137344, + "epoch": 6.893045002922268, + "grad_norm": 0.5807433128356934, + "learning_rate": 0.00012505671343755173, + "loss": 0.1908321762084961, + "mean_token_accuracy": 0.9318096882104874, + "num_tokens": 9134134.0, + "step": 2950 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4395190301266583, + "eval_loss": 0.9185124635696411, + "eval_mean_token_accuracy": 0.7936263409527865, + "eval_num_tokens": 9279501.0, + "eval_runtime": 37.9924, + "eval_samples_per_second": 32.401, + "eval_steps_per_second": 4.053, + "step": 2996 + } + ], + "logging_steps": 50, + "max_steps": 4280, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.3912420998060749e+17, + "train_batch_size": 4, + "trial_name": null, + "trial_params": null +} diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/README.md b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/adapter_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..0c2a7efe7af0ceadd86d0f851dc87d465191c777 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.04180832058159435, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "o_proj", + "q_proj", + "gate_proj", + "k_proj", + "down_proj", + "up_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/chat_template.jinja b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/tokenizer_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/trainer_state.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..fba14d8240b2820ddf6423b6d01f4d958a787940 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3424/trainer_state.json @@ -0,0 +1,802 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 8.0, + "eval_steps": 500, + "global_step": 3424, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8341716039180755, + "epoch": 0.11689070718877849, + "grad_norm": 1.1992322206497192, + "learning_rate": 5.366728168873013e-05, + "loss": 1.6918942260742187, + "mean_token_accuracy": 0.631798365265131, + "num_tokens": 153689.0, + "step": 50 + }, + { + "entropy": 0.9513179230690002, + "epoch": 0.23378141437755698, + "grad_norm": 1.1290454864501953, + "learning_rate": 0.00010842981402416902, + "loss": 0.8867578887939453, + "mean_token_accuracy": 0.7587438315153122, + "num_tokens": 310548.0, + "step": 100 + }, + { + "entropy": 0.8510265004634857, + "epoch": 0.3506721215663355, + "grad_norm": 0.7732136845588684, + "learning_rate": 0.00016319234635960792, + "loss": 0.7856448364257812, + "mean_token_accuracy": 0.7769897204637527, + "num_tokens": 468560.0, + "step": 150 + }, + { + "entropy": 0.8142367601394653, + "epoch": 0.46756282875511396, + "grad_norm": 0.7580806016921997, + "learning_rate": 0.00021795487869504684, + "loss": 0.7461666870117187, + "mean_token_accuracy": 0.7888966089487076, + "num_tokens": 619552.0, + "step": 200 + }, + { + "entropy": 0.772578022480011, + "epoch": 0.5844535359438925, + "grad_norm": 0.5330358147621155, + "learning_rate": 0.00027271741103048575, + "loss": 0.7142900848388671, + "mean_token_accuracy": 0.7959607627987861, + "num_tokens": 771972.0, + "step": 250 + }, + { + "entropy": 0.7481484657526016, + "epoch": 0.701344243132671, + "grad_norm": 0.8242517709732056, + "learning_rate": 0.00032747994336592464, + "loss": 0.6911422729492187, + "mean_token_accuracy": 0.8028205358982086, + "num_tokens": 928088.0, + "step": 300 + }, + { + "entropy": 0.741372903585434, + "epoch": 0.8182349503214494, + "grad_norm": 0.7722981572151184, + "learning_rate": 0.00038224247570136353, + "loss": 0.6908904266357422, + "mean_token_accuracy": 0.8018302822113037, + "num_tokens": 1086836.0, + "step": 350 + }, + { + "entropy": 0.7257227802276611, + "epoch": 0.9351256575102279, + "grad_norm": 0.7969573140144348, + "learning_rate": 0.0004370050080368024, + "loss": 0.6714310455322265, + "mean_token_accuracy": 0.8057891410589219, + "num_tokens": 1241227.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8317563653766334, + "eval_loss": 0.8135058283805847, + "eval_mean_token_accuracy": 0.7755894215850087, + "eval_num_tokens": 1325643.0, + "eval_runtime": 37.4951, + "eval_samples_per_second": 32.831, + "eval_steps_per_second": 4.107, + "step": 428 + }, + { + "entropy": 0.7179872094087265, + "epoch": 1.0514319111630626, + "grad_norm": 0.6224768161773682, + "learning_rate": 0.00046873290101955413, + "loss": 0.6683222961425781, + "mean_token_accuracy": 0.8054787133207273, + "num_tokens": 1393922.0, + "step": 450 + }, + { + "entropy": 0.6991294291615486, + "epoch": 1.1683226183518411, + "grad_norm": 0.6813111901283264, + "learning_rate": 0.00046837443306086947, + "loss": 0.6493977355957031, + "mean_token_accuracy": 0.8089349576830864, + "num_tokens": 1551442.0, + "step": 500 + }, + { + "entropy": 0.6951853144168854, + "epoch": 1.2852133255406195, + "grad_norm": 0.617091953754425, + "learning_rate": 0.0004676269147738558, + "loss": 0.6484123229980469, + "mean_token_accuracy": 0.8088837671279907, + "num_tokens": 1706697.0, + "step": 550 + }, + { + "entropy": 0.6852494943141937, + "epoch": 1.4021040327293979, + "grad_norm": 0.4677598774433136, + "learning_rate": 0.0004664915890374708, + "loss": 0.6416233062744141, + "mean_token_accuracy": 0.8106312158703805, + "num_tokens": 1865246.0, + "step": 600 + }, + { + "entropy": 0.6703594943881035, + "epoch": 1.5189947399181765, + "grad_norm": 0.5708025693893433, + "learning_rate": 0.0004649703435278991, + "loss": 0.6273183441162109, + "mean_token_accuracy": 0.8155005398392677, + "num_tokens": 2022060.0, + "step": 650 + }, + { + "entropy": 0.6676286320388317, + "epoch": 1.635885447106955, + "grad_norm": 0.5555347800254822, + "learning_rate": 0.00046306570757996264, + "loss": 0.6276700973510743, + "mean_token_accuracy": 0.8154828292131424, + "num_tokens": 2178909.0, + "step": 700 + }, + { + "entropy": 0.6669242936372757, + "epoch": 1.7527761542957334, + "grad_norm": 0.5260112285614014, + "learning_rate": 0.0004607808479816624, + "loss": 0.6228141784667969, + "mean_token_accuracy": 0.8164840793609619, + "num_tokens": 2329314.0, + "step": 750 + }, + { + "entropy": 0.6598560312390327, + "epoch": 1.869666861484512, + "grad_norm": 0.5677736401557922, + "learning_rate": 0.0004581195637088436, + "loss": 0.6214292907714843, + "mean_token_accuracy": 0.8171795177459716, + "num_tokens": 2478559.0, + "step": 800 + }, + { + "entropy": 0.6687870016694069, + "epoch": 1.9865575686732906, + "grad_norm": 0.6449615955352783, + "learning_rate": 0.00045508627960873823, + "loss": 0.6243909454345703, + "mean_token_accuracy": 0.8147780740261078, + "num_tokens": 2631937.0, + "step": 850 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7786144471013701, + "eval_loss": 0.7599140405654907, + "eval_mean_token_accuracy": 0.7935443106409791, + "eval_num_tokens": 2651286.0, + "eval_runtime": 39.0048, + "eval_samples_per_second": 31.56, + "eval_steps_per_second": 3.948, + "step": 856 + }, + { + "entropy": 0.5821921329701966, + "epoch": 2.102863822326125, + "grad_norm": 0.6116960048675537, + "learning_rate": 0.00045168603904288863, + "loss": 0.5406004714965821, + "mean_token_accuracy": 0.8334921918921734, + "num_tokens": 2795200.0, + "step": 900 + }, + { + "entropy": 0.5937331764400006, + "epoch": 2.2197545295149035, + "grad_norm": 0.5440122485160828, + "learning_rate": 0.00044792449550168286, + "loss": 0.5515737533569336, + "mean_token_accuracy": 0.8314691257476806, + "num_tokens": 2950534.0, + "step": 950 + }, + { + "entropy": 0.5877273553609847, + "epoch": 2.3366452367036823, + "grad_norm": 0.6505260467529297, + "learning_rate": 0.0004438079032044453, + "loss": 0.5507744979858399, + "mean_token_accuracy": 0.8314063146710395, + "num_tokens": 3101471.0, + "step": 1000 + }, + { + "entropy": 0.5897100016474723, + "epoch": 2.4535359438924607, + "grad_norm": 0.54567551612854, + "learning_rate": 0.0004393431067007111, + "loss": 0.5528768157958984, + "mean_token_accuracy": 0.8315245220065117, + "num_tokens": 3258554.0, + "step": 1050 + }, + { + "entropy": 0.5831590622663498, + "epoch": 2.570426651081239, + "grad_norm": 0.44826796650886536, + "learning_rate": 0.00043453752948997376, + "loss": 0.5442767333984375, + "mean_token_accuracy": 0.8327079233527184, + "num_tokens": 3416088.0, + "step": 1100 + }, + { + "entropy": 0.5839906217157841, + "epoch": 2.6873173582700174, + "grad_norm": 0.5067696571350098, + "learning_rate": 0.0004293991616788285, + "loss": 0.5485663604736328, + "mean_token_accuracy": 0.8324281191825866, + "num_tokens": 3567251.0, + "step": 1150 + }, + { + "entropy": 0.5870274990797043, + "epoch": 2.8042080654587958, + "grad_norm": 0.44032806158065796, + "learning_rate": 0.00042393654669603217, + "loss": 0.5524474716186524, + "mean_token_accuracy": 0.8313613015413285, + "num_tokens": 3719464.0, + "step": 1200 + }, + { + "entropy": 0.5864466108381748, + "epoch": 2.9210987726475746, + "grad_norm": 0.5520864129066467, + "learning_rate": 0.00041815876708756964, + "loss": 0.5490786361694336, + "mean_token_accuracy": 0.8314764249324799, + "num_tokens": 3874423.0, + "step": 1250 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6819627613990338, + "eval_loss": 0.7593190670013428, + "eval_mean_token_accuracy": 0.7857603395914102, + "eval_num_tokens": 3976929.0, + "eval_runtime": 39.3654, + "eval_samples_per_second": 31.271, + "eval_steps_per_second": 3.912, + "step": 1284 + }, + { + "entropy": 0.5587641523411525, + "epoch": 3.037405026300409, + "grad_norm": 0.622401773929596, + "learning_rate": 0.0004120754294153441, + "loss": 0.518932762145996, + "mean_token_accuracy": 0.8399260457436643, + "num_tokens": 4027609.0, + "step": 1300 + }, + { + "entropy": 0.5028628017008304, + "epoch": 3.1542957334891875, + "grad_norm": 0.577396810054779, + "learning_rate": 0.00040569664828459917, + "loss": 0.4589382171630859, + "mean_token_accuracy": 0.8538418188691139, + "num_tokens": 4180560.0, + "step": 1350 + }, + { + "entropy": 0.5227407096326351, + "epoch": 3.2711864406779663, + "grad_norm": 0.5113071203231812, + "learning_rate": 0.00039903302952663176, + "loss": 0.47863777160644533, + "mean_token_accuracy": 0.847364938557148, + "num_tokens": 4329954.0, + "step": 1400 + }, + { + "entropy": 0.5050077450275421, + "epoch": 3.3880771478667446, + "grad_norm": 0.48977184295654297, + "learning_rate": 0.0003920956525647558, + "loss": 0.46841480255126955, + "mean_token_accuracy": 0.8519425508379936, + "num_tokens": 4484354.0, + "step": 1450 + }, + { + "entropy": 0.5163291451334954, + "epoch": 3.504967855055523, + "grad_norm": 0.509864330291748, + "learning_rate": 0.000384896051992837, + "loss": 0.47587432861328127, + "mean_token_accuracy": 0.8486990982294083, + "num_tokens": 4638919.0, + "step": 1500 + }, + { + "entropy": 0.5134807989001274, + "epoch": 3.6218585622443014, + "grad_norm": 0.46544620394706726, + "learning_rate": 0.00037744619839702735, + "loss": 0.47692710876464844, + "mean_token_accuracy": 0.8489899519085884, + "num_tokens": 4795540.0, + "step": 1550 + }, + { + "entropy": 0.5216251534223556, + "epoch": 3.73874926943308, + "grad_norm": 0.5167004466056824, + "learning_rate": 0.0003697584784525874, + "loss": 0.4837848663330078, + "mean_token_accuracy": 0.847066233754158, + "num_tokens": 4951177.0, + "step": 1600 + }, + { + "entropy": 0.5254101701080799, + "epoch": 3.8556399766218585, + "grad_norm": 0.5504006743431091, + "learning_rate": 0.00036184567432888745, + "loss": 0.48759506225585936, + "mean_token_accuracy": 0.8457375919818878, + "num_tokens": 5104814.0, + "step": 1650 + }, + { + "entropy": 0.5107753933966159, + "epoch": 3.972530683810637, + "grad_norm": 0.421975314617157, + "learning_rate": 0.0003537209424368311, + "loss": 0.4743759536743164, + "mean_token_accuracy": 0.8505110186338425, + "num_tokens": 5265185.0, + "step": 1700 + }, + { + "epoch": 4.0, + "eval_entropy": 0.665543915002377, + "eval_loss": 0.7487082481384277, + "eval_mean_token_accuracy": 0.7999678904359991, + "eval_num_tokens": 5302572.0, + "eval_runtime": 38.5439, + "eval_samples_per_second": 31.938, + "eval_steps_per_second": 3.995, + "step": 1712 + }, + { + "entropy": 0.4414465431891494, + "epoch": 4.088836937463472, + "grad_norm": 0.5308493971824646, + "learning_rate": 0.0003453977915540383, + "loss": 0.39650299072265627, + "mean_token_accuracy": 0.8703725772287378, + "num_tokens": 5419002.0, + "step": 1750 + }, + { + "entropy": 0.42591195791959763, + "epoch": 4.20572764465225, + "grad_norm": 0.6111563444137573, + "learning_rate": 0.00033689006036415585, + "loss": 0.37838283538818357, + "mean_token_accuracy": 0.874235480427742, + "num_tokens": 5572969.0, + "step": 1800 + }, + { + "entropy": 0.44086351931095125, + "epoch": 4.322618351841029, + "grad_norm": 0.5524880290031433, + "learning_rate": 0.0003282118944476435, + "loss": 0.3920352554321289, + "mean_token_accuracy": 0.8694414687156677, + "num_tokens": 5733805.0, + "step": 1850 + }, + { + "entropy": 0.4380896310508251, + "epoch": 4.439509059029807, + "grad_norm": 0.5023863315582275, + "learning_rate": 0.0003193777227622898, + "loss": 0.3916081237792969, + "mean_token_accuracy": 0.8697139009833336, + "num_tokens": 5888694.0, + "step": 1900 + }, + { + "entropy": 0.44005885019898416, + "epoch": 4.556399766218585, + "grad_norm": 0.455997109413147, + "learning_rate": 0.000310402233652564, + "loss": 0.3955466842651367, + "mean_token_accuracy": 0.8693471103906631, + "num_tokens": 6042655.0, + "step": 1950 + }, + { + "entropy": 0.444269048422575, + "epoch": 4.673290473407365, + "grad_norm": 0.4542011320590973, + "learning_rate": 0.00030130035042769316, + "loss": 0.3988466262817383, + "mean_token_accuracy": 0.8680117425322532, + "num_tokens": 6203980.0, + "step": 2000 + }, + { + "entropy": 0.43520899042487143, + "epoch": 4.790181180596143, + "grad_norm": 0.49051445722579956, + "learning_rate": 0.0002920872065490688, + "loss": 0.39819797515869143, + "mean_token_accuracy": 0.8678940117359162, + "num_tokens": 6354604.0, + "step": 2050 + }, + { + "entropy": 0.429456724524498, + "epoch": 4.907071887784921, + "grad_norm": 0.5267980098724365, + "learning_rate": 0.0002827781204682396, + "loss": 0.39413917541503907, + "mean_token_accuracy": 0.870051506459713, + "num_tokens": 6507484.0, + "step": 2100 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5826305385146823, + "eval_loss": 0.7694042921066284, + "eval_mean_token_accuracy": 0.7950637824349589, + "eval_num_tokens": 6628215.0, + "eval_runtime": 38.5641, + "eval_samples_per_second": 31.921, + "eval_steps_per_second": 3.993, + "step": 2140 + }, + { + "entropy": 0.41423329688496324, + "epoch": 5.023378141437756, + "grad_norm": 0.6393507719039917, + "learning_rate": 0.0002733885701573256, + "loss": 0.3733938598632813, + "mean_token_accuracy": 0.8762160557598325, + "num_tokens": 6662015.0, + "step": 2150 + }, + { + "entropy": 0.32116023637354374, + "epoch": 5.140268848626534, + "grad_norm": 0.5128306150436401, + "learning_rate": 0.00026393416737420213, + "loss": 0.2751211166381836, + "mean_token_accuracy": 0.9052674892544746, + "num_tokens": 6817989.0, + "step": 2200 + }, + { + "entropy": 0.33186347484588624, + "epoch": 5.257159555815313, + "grad_norm": 0.6007080078125, + "learning_rate": 0.0002544306317052408, + "loss": 0.2850212097167969, + "mean_token_accuracy": 0.9016379952430725, + "num_tokens": 6977021.0, + "step": 2250 + }, + { + "entropy": 0.3474555689096451, + "epoch": 5.374050263004091, + "grad_norm": 0.6231185793876648, + "learning_rate": 0.0002448937644287679, + "loss": 0.29555959701538087, + "mean_token_accuracy": 0.8966628012061119, + "num_tokens": 7133360.0, + "step": 2300 + }, + { + "entropy": 0.346554354429245, + "epoch": 5.490940970192869, + "grad_norm": 0.5146846175193787, + "learning_rate": 0.0002353394222426952, + "loss": 0.29524572372436525, + "mean_token_accuracy": 0.8978548383712769, + "num_tokens": 7282372.0, + "step": 2350 + }, + { + "entropy": 0.3467242659628391, + "epoch": 5.607831677381649, + "grad_norm": 0.6118131875991821, + "learning_rate": 0.00022578349090000624, + "loss": 0.2970552635192871, + "mean_token_accuracy": 0.8985717830061912, + "num_tokens": 7435766.0, + "step": 2400 + }, + { + "entropy": 0.34968974225223065, + "epoch": 5.724722384570427, + "grad_norm": 0.6346496939659119, + "learning_rate": 0.0002162418587959333, + "loss": 0.29940626144409177, + "mean_token_accuracy": 0.8972864350676537, + "num_tokens": 7578408.0, + "step": 2450 + }, + { + "entropy": 0.35084108062088487, + "epoch": 5.841613091759205, + "grad_norm": 0.6566210389137268, + "learning_rate": 0.00020673039055074148, + "loss": 0.302426700592041, + "mean_token_accuracy": 0.8953999072313309, + "num_tokens": 7734599.0, + "step": 2500 + }, + { + "entropy": 0.33895430400967597, + "epoch": 5.958503798947984, + "grad_norm": 0.517574667930603, + "learning_rate": 0.00019726490063204247, + "loss": 0.292838020324707, + "mean_token_accuracy": 0.8984868070483207, + "num_tokens": 7898028.0, + "step": 2550 + }, + { + "epoch": 6.0, + "eval_entropy": 0.5211211553254684, + "eval_loss": 0.8269560933113098, + "eval_mean_token_accuracy": 0.7899065927251593, + "eval_num_tokens": 7953858.0, + "eval_runtime": 39.369, + "eval_samples_per_second": 31.268, + "eval_steps_per_second": 3.912, + "step": 2568 + }, + { + "entropy": 0.2763742347009218, + "epoch": 6.074810052600818, + "grad_norm": 0.5116816163063049, + "learning_rate": 0.00018786112706049623, + "loss": 0.22524795532226563, + "mean_token_accuracy": 0.9218554844209297, + "num_tokens": 8052646.0, + "step": 2600 + }, + { + "entropy": 0.23341606348752975, + "epoch": 6.1917007597895966, + "grad_norm": 0.6335962414741516, + "learning_rate": 0.00017853470524261842, + "loss": 0.18153846740722657, + "mean_token_accuracy": 0.935965863764286, + "num_tokens": 8212323.0, + "step": 2650 + }, + { + "entropy": 0.23515729174017908, + "epoch": 6.308591466978375, + "grad_norm": 0.6312636733055115, + "learning_rate": 0.0001693011419742025, + "loss": 0.18562854766845704, + "mean_token_accuracy": 0.9342144966125489, + "num_tokens": 8366715.0, + "step": 2700 + }, + { + "entropy": 0.23566059060394765, + "epoch": 6.425482174167154, + "grad_norm": 0.5869485139846802, + "learning_rate": 0.0001601757896575784, + "loss": 0.1867989158630371, + "mean_token_accuracy": 0.934665755033493, + "num_tokens": 8515955.0, + "step": 2750 + }, + { + "entropy": 0.2446490554511547, + "epoch": 6.5423728813559325, + "grad_norm": 0.5957017540931702, + "learning_rate": 0.00015117382077557817, + "loss": 0.19312274932861329, + "mean_token_accuracy": 0.9308532625436783, + "num_tokens": 8672361.0, + "step": 2800 + }, + { + "entropy": 0.24066772796213626, + "epoch": 6.659263588544711, + "grad_norm": 0.6069761514663696, + "learning_rate": 0.0001423102026646483, + "loss": 0.19540096282958985, + "mean_token_accuracy": 0.9300534284114838, + "num_tokens": 8827819.0, + "step": 2850 + }, + { + "entropy": 0.2378876845538616, + "epoch": 6.776154295733489, + "grad_norm": 0.6069548726081848, + "learning_rate": 0.00013359967262905405, + "loss": 0.1942250633239746, + "mean_token_accuracy": 0.9310428243875504, + "num_tokens": 8979535.0, + "step": 2900 + }, + { + "entropy": 0.23810458809137344, + "epoch": 6.893045002922268, + "grad_norm": 0.5807433128356934, + "learning_rate": 0.00012505671343755173, + "loss": 0.1908321762084961, + "mean_token_accuracy": 0.9318096882104874, + "num_tokens": 9134134.0, + "step": 2950 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4395190301266583, + "eval_loss": 0.9185124635696411, + "eval_mean_token_accuracy": 0.7936263409527865, + "eval_num_tokens": 9279501.0, + "eval_runtime": 37.9924, + "eval_samples_per_second": 32.401, + "eval_steps_per_second": 4.053, + "step": 2996 + }, + { + "entropy": 0.22885651856511083, + "epoch": 7.009351256575102, + "grad_norm": 0.49437037110328674, + "learning_rate": 0.00011669552924327112, + "loss": 0.18458213806152343, + "mean_token_accuracy": 0.9342338658457425, + "num_tokens": 9290416.0, + "step": 3000 + }, + { + "entropy": 0.1516158328205347, + "epoch": 7.1262419637638805, + "grad_norm": 0.5319057703018188, + "learning_rate": 0.00010853002196684399, + "loss": 0.10484781265258789, + "mean_token_accuracy": 0.9638857007026672, + "num_tokens": 9444705.0, + "step": 3050 + }, + { + "entropy": 0.14421835087239743, + "epoch": 7.243132670952659, + "grad_norm": 0.7959536910057068, + "learning_rate": 0.00010057376818204711, + "loss": 0.10359338760375976, + "mean_token_accuracy": 0.964658388197422, + "num_tokens": 9601476.0, + "step": 3100 + }, + { + "entropy": 0.14138914909213782, + "epoch": 7.360023378141438, + "grad_norm": 0.5152567028999329, + "learning_rate": 9.283999654239098e-05, + "loss": 0.10241118431091309, + "mean_token_accuracy": 0.9644565668702125, + "num_tokens": 9757913.0, + "step": 3150 + }, + { + "entropy": 0.14779060527682306, + "epoch": 7.4769140853302165, + "grad_norm": 0.5124489068984985, + "learning_rate": 8.534156578618579e-05, + "loss": 0.10744266510009766, + "mean_token_accuracy": 0.9628658410906792, + "num_tokens": 9905790.0, + "step": 3200 + }, + { + "entropy": 0.1453312272951007, + "epoch": 7.593804792518995, + "grad_norm": 0.4871680438518524, + "learning_rate": 7.809094335665658e-05, + "loss": 0.10566988945007325, + "mean_token_accuracy": 0.9635729387402534, + "num_tokens": 10064234.0, + "step": 3250 + }, + { + "entropy": 0.14601945489645005, + "epoch": 7.710695499707773, + "grad_norm": 0.5538403987884521, + "learning_rate": 7.110018467265364e-05, + "loss": 0.10567939758300782, + "mean_token_accuracy": 0.9625948050618172, + "num_tokens": 10221089.0, + "step": 3300 + }, + { + "entropy": 0.14106836922466756, + "epoch": 7.827586206896552, + "grad_norm": 0.4848695695400238, + "learning_rate": 6.438091308442527e-05, + "loss": 0.10246496200561524, + "mean_token_accuracy": 0.9643726572394371, + "num_tokens": 10379655.0, + "step": 3350 + }, + { + "entropy": 0.14235207289457322, + "epoch": 7.94447691408533, + "grad_norm": 0.5310181379318237, + "learning_rate": 5.7944300547779826e-05, + "loss": 0.10354788780212403, + "mean_token_accuracy": 0.9633960571885108, + "num_tokens": 10531301.0, + "step": 3400 + }, + { + "epoch": 8.0, + "eval_entropy": 0.3818505619253431, + "eval_loss": 1.0742957592010498, + "eval_mean_token_accuracy": 0.7910223340059256, + "eval_num_tokens": 10605144.0, + "eval_runtime": 38.0708, + "eval_samples_per_second": 32.335, + "eval_steps_per_second": 4.045, + "step": 3424 + } + ], + "logging_steps": 50, + "max_steps": 4280, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.589156466300795e+17, + "train_batch_size": 4, + "trial_name": null, + "trial_params": null +} diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/README.md b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/adapter_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..0c2a7efe7af0ceadd86d0f851dc87d465191c777 --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-3852/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.04180832058159435, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "o_proj", + "q_proj", + "gate_proj", + "k_proj", + "down_proj", + "up_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-4280/tokenizer_config.json b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-4280/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/DBCA_original_Estonian/Qwen3.5-2B-Base_original_features_structural_train_original_features_structural_test1/checkpoint-4280/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/README.md new file mode 100644 index 0000000000000000000000000000000000000000..dd5a19f4b6265236f518255cb787811724b3becf --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/README.md @@ -0,0 +1,58 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: transformers +model_name: Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1 +tags: +- generated_from_trainer +- sft +- trl +licence: license +--- + +# Model Card for Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1 + +This model is a fine-tuned version of [Qwen/Qwen3.5-2B-Base](https://huggingface.co/Qwen/Qwen3.5-2B-Base). +It has been trained using [TRL](https://github.com/huggingface/trl). + +## Quick start + +```python +from transformers import pipeline + +question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?" +generator = pipeline("text-generation", model="None", device="cuda") +output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0] +print(output["generated_text"]) +``` + +## Training procedure + +[Visualize in Weights & Biases](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/9asazldh) + + + +This model was trained with SFT. + +### Framework versions + +- TRL: 0.29.0 +- Transformers: 5.5.4 +- Pytorch: 2.10.0 +- Datasets: 4.6.1 +- Tokenizers: 0.22.2 + +## Citations + + + +Cite TRL as: + +```bibtex +@software{vonwerra2020trl, + title = {{TRL: Transformers Reinforcement Learning}}, + author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin}, + license = {Apache-2.0}, + url = {https://github.com/huggingface/trl}, + year = {2020} +} +``` \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b9596096c71017d49955136a4e23d9f97138d305 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.08602048083239339, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9106ebb115e5e5b8f52f77133ccdd55d333f4788 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1224/trainer_state.json @@ -0,0 +1,340 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 3.0, + "eval_steps": 500, + "global_step": 1224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 2.2408065378665922, + "epoch": 0.12269938650306748, + "grad_norm": 1.6718658208847046, + "learning_rate": 4.487804970912036e-05, + "loss": 2.129624786376953, + "mean_token_accuracy": 0.5743644836544991, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 1.1078504550457, + "epoch": 0.24539877300613497, + "grad_norm": 2.4091813564300537, + "learning_rate": 9.067197798373296e-05, + "loss": 1.0346390533447265, + "mean_token_accuracy": 0.7085381114482879, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.9421598988771439, + "epoch": 0.36809815950920244, + "grad_norm": 1.3913646936416626, + "learning_rate": 0.00013646590625834558, + "loss": 0.8716551208496094, + "mean_token_accuracy": 0.7449040985107422, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8723550146818161, + "epoch": 0.49079754601226994, + "grad_norm": 1.1664468050003052, + "learning_rate": 0.0001822598345329582, + "loss": 0.8095977783203125, + "mean_token_accuracy": 0.7585240352153778, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8466519421339035, + "epoch": 0.6134969325153374, + "grad_norm": 1.0145421028137207, + "learning_rate": 0.00022805376280757083, + "loss": 0.7830857849121093, + "mean_token_accuracy": 0.7661705195903779, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.8180295366048813, + "epoch": 0.7361963190184049, + "grad_norm": 0.9731705784797668, + "learning_rate": 0.00027384769108218336, + "loss": 0.7573603820800782, + "mean_token_accuracy": 0.7704453605413437, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7990227991342544, + "epoch": 0.8588957055214724, + "grad_norm": 1.0642374753952026, + "learning_rate": 0.000319641619356796, + "loss": 0.73709228515625, + "mean_token_accuracy": 0.7752733880281448, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7899087685346603, + "epoch": 0.9815950920245399, + "grad_norm": 0.880613386631012, + "learning_rate": 0.0003654355476314086, + "loss": 0.7347319793701171, + "mean_token_accuracy": 0.7766731631755829, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.83522472347532, + "eval_mean_token_accuracy": 0.7484098076820374, + "eval_not_syn_loss": 0.7872886061668396, + "eval_not_syn_runtime": 54.8697, + "eval_not_syn_samples_per_second": 25.424, + "eval_not_syn_steps_per_second": 3.189, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7784442271505083, + "eval_mean_token_accuracy": 0.8021791185651507, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.6980098485946655, + "eval_syn_runtime": 56.6342, + "eval_syn_samples_per_second": 24.632, + "eval_syn_steps_per_second": 3.09, + "step": 408 + }, + { + "entropy": 0.75672573694075, + "epoch": 1.1030674846625768, + "grad_norm": 0.9244198203086853, + "learning_rate": 0.00037356351883456205, + "loss": 0.7048745727539063, + "mean_token_accuracy": 0.783016896609104, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7406797379255294, + "epoch": 1.2257668711656442, + "grad_norm": 0.6744896769523621, + "learning_rate": 0.00037311248151765587, + "loss": 0.6860730743408203, + "mean_token_accuracy": 0.787369327545166, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7424973237514496, + "epoch": 1.3484662576687116, + "grad_norm": 0.92305588722229, + "learning_rate": 0.000372320629215228, + "loss": 0.6855724334716797, + "mean_token_accuracy": 0.7871134179830551, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7502243828773498, + "epoch": 1.471165644171779, + "grad_norm": 0.8259047269821167, + "learning_rate": 0.00037118941074037944, + "loss": 0.6949834442138672, + "mean_token_accuracy": 0.7851914083957672, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7233260518312454, + "epoch": 1.5938650306748468, + "grad_norm": 0.882722795009613, + "learning_rate": 0.00036972089582775814, + "loss": 0.6649005889892579, + "mean_token_accuracy": 0.7898879665136337, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7162708967924118, + "epoch": 1.716564417177914, + "grad_norm": 0.6593027114868164, + "learning_rate": 0.0003679177713466678, + "loss": 0.6653710174560546, + "mean_token_accuracy": 0.7905678844451904, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7150520020723343, + "epoch": 1.8392638036809816, + "grad_norm": 0.665340781211853, + "learning_rate": 0.0003657833363850354, + "loss": 0.6600938415527344, + "mean_token_accuracy": 0.7918649202585221, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.6993386310338974, + "epoch": 1.961963190184049, + "grad_norm": 0.6502187252044678, + "learning_rate": 0.00036332149621323294, + "loss": 0.6469607543945313, + "mean_token_accuracy": 0.7950778317451477, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7315139893123082, + "eval_mean_token_accuracy": 0.7713504004478454, + "eval_not_syn_loss": 0.699893593788147, + "eval_not_syn_runtime": 55.4135, + "eval_not_syn_samples_per_second": 25.174, + "eval_not_syn_steps_per_second": 3.158, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.6848451798302787, + "eval_mean_token_accuracy": 0.814895657471248, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6415576934814453, + "eval_syn_runtime": 56.4735, + "eval_syn_samples_per_second": 24.702, + "eval_syn_steps_per_second": 3.099, + "step": 816 + }, + { + "entropy": 0.6596363644407253, + "epoch": 2.083435582822086, + "grad_norm": 0.6647700071334839, + "learning_rate": 0.00036053675513879666, + "loss": 0.599151611328125, + "mean_token_accuracy": 0.8071986864311527, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6626429110765457, + "epoch": 2.2061349693251535, + "grad_norm": 0.7356868386268616, + "learning_rate": 0.00035743420826511783, + "loss": 0.6004195404052735, + "mean_token_accuracy": 0.805362731218338, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6503470712900161, + "epoch": 2.3288343558282207, + "grad_norm": 1.095608115196228, + "learning_rate": 0.0003540195321691835, + "loss": 0.5951333999633789, + "mean_token_accuracy": 0.807324025630951, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.660577870607376, + "epoch": 2.4515337423312884, + "grad_norm": 0.6602179408073425, + "learning_rate": 0.00035029897451542325, + "loss": 0.6056365585327148, + "mean_token_accuracy": 0.8057438349723816, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6555853113532066, + "epoch": 2.574233128834356, + "grad_norm": 0.7459151148796082, + "learning_rate": 0.00034627934262466606, + "loss": 0.5963204193115235, + "mean_token_accuracy": 0.8060598099231719, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6505853188037872, + "epoch": 2.6969325153374233, + "grad_norm": 0.6359792947769165, + "learning_rate": 0.00034196799101912103, + "loss": 0.5993065643310547, + "mean_token_accuracy": 0.806280387043953, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.6509970253705979, + "epoch": 2.819631901840491, + "grad_norm": 0.8086174726486206, + "learning_rate": 0.0003373728079661707, + "loss": 0.6001268005371094, + "mean_token_accuracy": 0.806866540312767, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6404844307899475, + "epoch": 2.942331288343558, + "grad_norm": 0.7038407325744629, + "learning_rate": 0.0003325022010455975, + "loss": 0.5870526123046875, + "mean_token_accuracy": 0.8104586935043335, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6645996727262224, + "eval_mean_token_accuracy": 0.798654237134116, + "eval_not_syn_loss": 0.6586771011352539, + "eval_not_syn_runtime": 55.1088, + "eval_not_syn_samples_per_second": 25.314, + "eval_not_syn_steps_per_second": 3.176, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6233607809884207, + "eval_mean_token_accuracy": 0.7977249867575509, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6288298964500427, + "eval_syn_runtime": 56.6081, + "eval_syn_samples_per_second": 24.643, + "eval_syn_steps_per_second": 3.091, + "step": 1224 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.458126379206016e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b9596096c71017d49955136a4e23d9f97138d305 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test1/checkpoint-1632/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.08602048083239339, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8e55572239b6c32b5b9b6754aefce409039ac9ab --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/README.md @@ -0,0 +1,58 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: transformers +model_name: Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2 +tags: +- generated_from_trainer +- sft +- trl +licence: license +--- + +# Model Card for Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2 + +This model is a fine-tuned version of [Qwen/Qwen3.5-2B-Base](https://huggingface.co/Qwen/Qwen3.5-2B-Base). +It has been trained using [TRL](https://github.com/huggingface/trl). + +## Quick start + +```python +from transformers import pipeline + +question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?" +generator = pipeline("text-generation", model="None", device="cuda") +output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0] +print(output["generated_text"]) +``` + +## Training procedure + +[Visualize in Weights & Biases](https://wandb.ai/katriin-kukk/Cross_lingual_morphological_generalization/runs/ypap1lsm) + + + +This model was trained with SFT. + +### Framework versions + +- TRL: 0.29.0 +- Transformers: 5.5.4 +- Pytorch: 2.10.0 +- Datasets: 4.6.1 +- Tokenizers: 0.22.2 + +## Citations + + + +Cite TRL as: + +```bibtex +@software{vonwerra2020trl, + title = {{TRL: Transformers Reinforcement Learning}}, + author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin}, + license = {Apache-2.0}, + url = {https://github.com/huggingface/trl}, + year = {2020} +} +``` \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c75693c13b7b0250de0b2dc43ad401b8b00c19fc --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1224/trainer_state.json @@ -0,0 +1,340 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 3.0, + "eval_steps": 500, + "global_step": 1224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.610858430973312e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6dc70f3599903b9f071a73ade442f5211ddadc5a --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-1632/trainer_state.json @@ -0,0 +1,442 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 4.0, + "eval_steps": 500, + "global_step": 1632, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.819362572820186e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ecdef5c4a11bc3deb327cd7941d115ac5fa346a2 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2040/trainer_state.json @@ -0,0 +1,544 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 5.0, + "eval_steps": 500, + "global_step": 2040, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.1014210793408448e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bd633ec5467130b9f20492716e1001f23958ba49 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2448/trainer_state.json @@ -0,0 +1,646 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 6.0, + "eval_steps": 500, + "global_step": 2448, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + }, + { + "entropy": 0.4960306563762703, + "epoch": 5.024539877300613, + "grad_norm": 0.9411171078681946, + "learning_rate": 0.0002353013459391488, + "loss": 0.4484004211425781, + "mean_token_accuracy": 0.8438825571175778, + "num_tokens": 5427362.0, + "step": 2050 + }, + { + "entropy": 0.3916636416316032, + "epoch": 5.147239263803681, + "grad_norm": 0.8929405808448792, + "learning_rate": 0.00022676095505345462, + "loss": 0.3446478271484375, + "mean_token_accuracy": 0.8732858788967133, + "num_tokens": 5564186.0, + "step": 2100 + }, + { + "entropy": 0.39636621803045274, + "epoch": 5.269938650306749, + "grad_norm": 0.69569331407547, + "learning_rate": 0.0002181748793031656, + "loss": 0.3548049163818359, + "mean_token_accuracy": 0.8699262255430221, + "num_tokens": 5694728.0, + "step": 2150 + }, + { + "entropy": 0.4028259950876236, + "epoch": 5.392638036809816, + "grad_norm": 0.7605823874473572, + "learning_rate": 0.00020955882820758868, + "loss": 0.35218334197998047, + "mean_token_accuracy": 0.8704028391838073, + "num_tokens": 5832581.0, + "step": 2200 + }, + { + "entropy": 0.4091260200738907, + "epoch": 5.515337423312883, + "grad_norm": 0.8990920782089233, + "learning_rate": 0.00020092856613044126, + "loss": 0.36372703552246094, + "mean_token_accuracy": 0.8663889598846436, + "num_tokens": 5962417.0, + "step": 2250 + }, + { + "entropy": 0.4057882487773895, + "epoch": 5.638036809815951, + "grad_norm": 0.9872731566429138, + "learning_rate": 0.0001922998834365732, + "loss": 0.3588364791870117, + "mean_token_accuracy": 0.8686526268720627, + "num_tokens": 6092338.0, + "step": 2300 + }, + { + "entropy": 0.41128933161497117, + "epoch": 5.7607361963190185, + "grad_norm": 0.8626382350921631, + "learning_rate": 0.0001836885676011143, + "loss": 0.36508998870849607, + "mean_token_accuracy": 0.8664334756135941, + "num_tokens": 6226327.0, + "step": 2350 + }, + { + "entropy": 0.40721403241157533, + "epoch": 5.883435582822086, + "grad_norm": 0.775476336479187, + "learning_rate": 0.00017511037432391027, + "loss": 0.36363380432128906, + "mean_token_accuracy": 0.8656364333629608, + "num_tokens": 6359083.0, + "step": 2400 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4985327124595642, + "eval_mean_token_accuracy": 0.8019610796655927, + "eval_not_syn_loss": 0.717173159122467, + "eval_not_syn_runtime": 56.4169, + "eval_not_syn_samples_per_second": 24.744, + "eval_not_syn_steps_per_second": 3.102, + "eval_num_tokens": 6481176.0, + "step": 2448 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4651233986445836, + "eval_mean_token_accuracy": 0.8059938202585493, + "eval_num_tokens": 6481176.0, + "eval_syn_loss": 0.6778863668441772, + "eval_syn_runtime": 58.5069, + "eval_syn_samples_per_second": 23.86, + "eval_syn_steps_per_second": 2.991, + "step": 2448 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.321195971768434e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..83c225fb8c90fe3f001ba06eed7a04abb105ff31 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-2856/trainer_state.json @@ -0,0 +1,758 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 7.0, + "eval_steps": 500, + "global_step": 2856, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + }, + { + "entropy": 0.4960306563762703, + "epoch": 5.024539877300613, + "grad_norm": 0.9411171078681946, + "learning_rate": 0.0002353013459391488, + "loss": 0.4484004211425781, + "mean_token_accuracy": 0.8438825571175778, + "num_tokens": 5427362.0, + "step": 2050 + }, + { + "entropy": 0.3916636416316032, + "epoch": 5.147239263803681, + "grad_norm": 0.8929405808448792, + "learning_rate": 0.00022676095505345462, + "loss": 0.3446478271484375, + "mean_token_accuracy": 0.8732858788967133, + "num_tokens": 5564186.0, + "step": 2100 + }, + { + "entropy": 0.39636621803045274, + "epoch": 5.269938650306749, + "grad_norm": 0.69569331407547, + "learning_rate": 0.0002181748793031656, + "loss": 0.3548049163818359, + "mean_token_accuracy": 0.8699262255430221, + "num_tokens": 5694728.0, + "step": 2150 + }, + { + "entropy": 0.4028259950876236, + "epoch": 5.392638036809816, + "grad_norm": 0.7605823874473572, + "learning_rate": 0.00020955882820758868, + "loss": 0.35218334197998047, + "mean_token_accuracy": 0.8704028391838073, + "num_tokens": 5832581.0, + "step": 2200 + }, + { + "entropy": 0.4091260200738907, + "epoch": 5.515337423312883, + "grad_norm": 0.8990920782089233, + "learning_rate": 0.00020092856613044126, + "loss": 0.36372703552246094, + "mean_token_accuracy": 0.8663889598846436, + "num_tokens": 5962417.0, + "step": 2250 + }, + { + "entropy": 0.4057882487773895, + "epoch": 5.638036809815951, + "grad_norm": 0.9872731566429138, + "learning_rate": 0.0001922998834365732, + "loss": 0.3588364791870117, + "mean_token_accuracy": 0.8686526268720627, + "num_tokens": 6092338.0, + "step": 2300 + }, + { + "entropy": 0.41128933161497117, + "epoch": 5.7607361963190185, + "grad_norm": 0.8626382350921631, + "learning_rate": 0.0001836885676011143, + "loss": 0.36508998870849607, + "mean_token_accuracy": 0.8664334756135941, + "num_tokens": 6226327.0, + "step": 2350 + }, + { + "entropy": 0.40721403241157533, + "epoch": 5.883435582822086, + "grad_norm": 0.775476336479187, + "learning_rate": 0.00017511037432391027, + "loss": 0.36363380432128906, + "mean_token_accuracy": 0.8656364333629608, + "num_tokens": 6359083.0, + "step": 2400 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4985327124595642, + "eval_mean_token_accuracy": 0.8019610796655927, + "eval_not_syn_loss": 0.717173159122467, + "eval_not_syn_runtime": 56.4169, + "eval_not_syn_samples_per_second": 24.744, + "eval_not_syn_steps_per_second": 3.102, + "eval_num_tokens": 6481176.0, + "step": 2448 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4651233986445836, + "eval_mean_token_accuracy": 0.8059938202585493, + "eval_num_tokens": 6481176.0, + "eval_syn_loss": 0.6778863668441772, + "eval_syn_runtime": 58.5069, + "eval_syn_samples_per_second": 23.86, + "eval_syn_steps_per_second": 2.991, + "step": 2448 + }, + { + "entropy": 0.40582828509687174, + "epoch": 6.004907975460123, + "grad_norm": 0.6788471341133118, + "learning_rate": 0.0001665809987020958, + "loss": 0.3566559600830078, + "mean_token_accuracy": 0.868294438930473, + "num_tokens": 6486549.0, + "step": 2450 + }, + { + "entropy": 0.2933592677116394, + "epoch": 6.12760736196319, + "grad_norm": 0.6598446369171143, + "learning_rate": 0.00015811604651354912, + "loss": 0.24192693710327148, + "mean_token_accuracy": 0.9082661873102188, + "num_tokens": 6622384.0, + "step": 2500 + }, + { + "entropy": 0.2975004023313522, + "epoch": 6.250306748466258, + "grad_norm": 0.7413120865821838, + "learning_rate": 0.00014973100566377046, + "loss": 0.2502699661254883, + "mean_token_accuracy": 0.9056042343378067, + "num_tokens": 6758637.0, + "step": 2550 + }, + { + "entropy": 0.3052474121749401, + "epoch": 6.373006134969325, + "grad_norm": 0.8629281520843506, + "learning_rate": 0.00014144121784842467, + "loss": 0.2543815040588379, + "mean_token_accuracy": 0.9046169209480286, + "num_tokens": 6888533.0, + "step": 2600 + }, + { + "entropy": 0.303673956990242, + "epoch": 6.495705521472392, + "grad_norm": 0.8680911660194397, + "learning_rate": 0.000133261850483398, + "loss": 0.2529561424255371, + "mean_token_accuracy": 0.9040863698720932, + "num_tokens": 7023729.0, + "step": 2650 + }, + { + "entropy": 0.2993183335661888, + "epoch": 6.61840490797546, + "grad_norm": 0.7648728489875793, + "learning_rate": 0.00012520786895372493, + "loss": 0.2517524528503418, + "mean_token_accuracy": 0.9038773095607757, + "num_tokens": 7153821.0, + "step": 2700 + }, + { + "entropy": 0.3006903246045113, + "epoch": 6.741104294478528, + "grad_norm": 0.7433948516845703, + "learning_rate": 0.00011729400923216141, + "loss": 0.25459712982177735, + "mean_token_accuracy": 0.9037652736902237, + "num_tokens": 7283513.0, + "step": 2750 + }, + { + "entropy": 0.30416361182928087, + "epoch": 6.863803680981595, + "grad_norm": 0.8744818568229675, + "learning_rate": 0.00010953475091750244, + "loss": 0.25621614456176756, + "mean_token_accuracy": 0.9019701254367828, + "num_tokens": 7411578.0, + "step": 2800 + }, + { + "entropy": 0.2956543755531311, + "epoch": 6.986503067484662, + "grad_norm": 0.8242411017417908, + "learning_rate": 0.00010194429074197415, + "loss": 0.24829059600830078, + "mean_token_accuracy": 0.9054388856887817, + "num_tokens": 7548356.0, + "step": 2850 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4070556444781167, + "eval_mean_token_accuracy": 0.7848251972879682, + "eval_not_syn_loss": 0.8054981827735901, + "eval_not_syn_runtime": 56.6822, + "eval_not_syn_samples_per_second": 24.629, + "eval_not_syn_steps_per_second": 3.087, + "eval_num_tokens": 7561372.0, + "step": 2856 + }, + { + "epoch": 7.0, + "eval_entropy": 0.3784746325016022, + "eval_mean_token_accuracy": 0.8244705755370004, + "eval_num_tokens": 7561372.0, + "eval_syn_loss": 0.7377892732620239, + "eval_syn_runtime": 58.6534, + "eval_syn_samples_per_second": 23.801, + "eval_syn_steps_per_second": 2.984, + "step": 2856 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.5400107074712845e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e3d680e4a4134f7178937f809d8ae398d9d4293d --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3264/trainer_state.json @@ -0,0 +1,860 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 8.0, + "eval_steps": 500, + "global_step": 3264, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + }, + { + "entropy": 0.4960306563762703, + "epoch": 5.024539877300613, + "grad_norm": 0.9411171078681946, + "learning_rate": 0.0002353013459391488, + "loss": 0.4484004211425781, + "mean_token_accuracy": 0.8438825571175778, + "num_tokens": 5427362.0, + "step": 2050 + }, + { + "entropy": 0.3916636416316032, + "epoch": 5.147239263803681, + "grad_norm": 0.8929405808448792, + "learning_rate": 0.00022676095505345462, + "loss": 0.3446478271484375, + "mean_token_accuracy": 0.8732858788967133, + "num_tokens": 5564186.0, + "step": 2100 + }, + { + "entropy": 0.39636621803045274, + "epoch": 5.269938650306749, + "grad_norm": 0.69569331407547, + "learning_rate": 0.0002181748793031656, + "loss": 0.3548049163818359, + "mean_token_accuracy": 0.8699262255430221, + "num_tokens": 5694728.0, + "step": 2150 + }, + { + "entropy": 0.4028259950876236, + "epoch": 5.392638036809816, + "grad_norm": 0.7605823874473572, + "learning_rate": 0.00020955882820758868, + "loss": 0.35218334197998047, + "mean_token_accuracy": 0.8704028391838073, + "num_tokens": 5832581.0, + "step": 2200 + }, + { + "entropy": 0.4091260200738907, + "epoch": 5.515337423312883, + "grad_norm": 0.8990920782089233, + "learning_rate": 0.00020092856613044126, + "loss": 0.36372703552246094, + "mean_token_accuracy": 0.8663889598846436, + "num_tokens": 5962417.0, + "step": 2250 + }, + { + "entropy": 0.4057882487773895, + "epoch": 5.638036809815951, + "grad_norm": 0.9872731566429138, + "learning_rate": 0.0001922998834365732, + "loss": 0.3588364791870117, + "mean_token_accuracy": 0.8686526268720627, + "num_tokens": 6092338.0, + "step": 2300 + }, + { + "entropy": 0.41128933161497117, + "epoch": 5.7607361963190185, + "grad_norm": 0.8626382350921631, + "learning_rate": 0.0001836885676011143, + "loss": 0.36508998870849607, + "mean_token_accuracy": 0.8664334756135941, + "num_tokens": 6226327.0, + "step": 2350 + }, + { + "entropy": 0.40721403241157533, + "epoch": 5.883435582822086, + "grad_norm": 0.775476336479187, + "learning_rate": 0.00017511037432391027, + "loss": 0.36363380432128906, + "mean_token_accuracy": 0.8656364333629608, + "num_tokens": 6359083.0, + "step": 2400 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4985327124595642, + "eval_mean_token_accuracy": 0.8019610796655927, + "eval_not_syn_loss": 0.717173159122467, + "eval_not_syn_runtime": 56.4169, + "eval_not_syn_samples_per_second": 24.744, + "eval_not_syn_steps_per_second": 3.102, + "eval_num_tokens": 6481176.0, + "step": 2448 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4651233986445836, + "eval_mean_token_accuracy": 0.8059938202585493, + "eval_num_tokens": 6481176.0, + "eval_syn_loss": 0.6778863668441772, + "eval_syn_runtime": 58.5069, + "eval_syn_samples_per_second": 23.86, + "eval_syn_steps_per_second": 2.991, + "step": 2448 + }, + { + "entropy": 0.40582828509687174, + "epoch": 6.004907975460123, + "grad_norm": 0.6788471341133118, + "learning_rate": 0.0001665809987020958, + "loss": 0.3566559600830078, + "mean_token_accuracy": 0.868294438930473, + "num_tokens": 6486549.0, + "step": 2450 + }, + { + "entropy": 0.2933592677116394, + "epoch": 6.12760736196319, + "grad_norm": 0.6598446369171143, + "learning_rate": 0.00015811604651354912, + "loss": 0.24192693710327148, + "mean_token_accuracy": 0.9082661873102188, + "num_tokens": 6622384.0, + "step": 2500 + }, + { + "entropy": 0.2975004023313522, + "epoch": 6.250306748466258, + "grad_norm": 0.7413120865821838, + "learning_rate": 0.00014973100566377046, + "loss": 0.2502699661254883, + "mean_token_accuracy": 0.9056042343378067, + "num_tokens": 6758637.0, + "step": 2550 + }, + { + "entropy": 0.3052474121749401, + "epoch": 6.373006134969325, + "grad_norm": 0.8629281520843506, + "learning_rate": 0.00014144121784842467, + "loss": 0.2543815040588379, + "mean_token_accuracy": 0.9046169209480286, + "num_tokens": 6888533.0, + "step": 2600 + }, + { + "entropy": 0.303673956990242, + "epoch": 6.495705521472392, + "grad_norm": 0.8680911660194397, + "learning_rate": 0.000133261850483398, + "loss": 0.2529561424255371, + "mean_token_accuracy": 0.9040863698720932, + "num_tokens": 7023729.0, + "step": 2650 + }, + { + "entropy": 0.2993183335661888, + "epoch": 6.61840490797546, + "grad_norm": 0.7648728489875793, + "learning_rate": 0.00012520786895372493, + "loss": 0.2517524528503418, + "mean_token_accuracy": 0.9038773095607757, + "num_tokens": 7153821.0, + "step": 2700 + }, + { + "entropy": 0.3006903246045113, + "epoch": 6.741104294478528, + "grad_norm": 0.7433948516845703, + "learning_rate": 0.00011729400923216141, + "loss": 0.25459712982177735, + "mean_token_accuracy": 0.9037652736902237, + "num_tokens": 7283513.0, + "step": 2750 + }, + { + "entropy": 0.30416361182928087, + "epoch": 6.863803680981595, + "grad_norm": 0.8744818568229675, + "learning_rate": 0.00010953475091750244, + "loss": 0.25621614456176756, + "mean_token_accuracy": 0.9019701254367828, + "num_tokens": 7411578.0, + "step": 2800 + }, + { + "entropy": 0.2956543755531311, + "epoch": 6.986503067484662, + "grad_norm": 0.8242411017417908, + "learning_rate": 0.00010194429074197415, + "loss": 0.24829059600830078, + "mean_token_accuracy": 0.9054388856887817, + "num_tokens": 7548356.0, + "step": 2850 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4070556444781167, + "eval_mean_token_accuracy": 0.7848251972879682, + "eval_not_syn_loss": 0.8054981827735901, + "eval_not_syn_runtime": 56.6822, + "eval_not_syn_samples_per_second": 24.629, + "eval_not_syn_steps_per_second": 3.087, + "eval_num_tokens": 7561372.0, + "step": 2856 + }, + { + "epoch": 7.0, + "eval_entropy": 0.3784746325016022, + "eval_mean_token_accuracy": 0.8244705755370004, + "eval_num_tokens": 7561372.0, + "eval_syn_loss": 0.7377892732620239, + "eval_syn_runtime": 58.6534, + "eval_syn_samples_per_second": 23.801, + "eval_syn_steps_per_second": 2.984, + "step": 2856 + }, + { + "entropy": 0.22089176542229122, + "epoch": 7.1079754601227, + "grad_norm": 0.7850068211555481, + "learning_rate": 9.453651659617315e-05, + "loss": 0.17431402206420898, + "mean_token_accuracy": 0.9356574006754943, + "num_tokens": 7671657.0, + "step": 2900 + }, + { + "entropy": 0.2014119729399681, + "epoch": 7.230674846625767, + "grad_norm": 0.6396872401237488, + "learning_rate": 8.732498211907838e-05, + "loss": 0.15425819396972656, + "mean_token_accuracy": 0.9421506607532502, + "num_tokens": 7806989.0, + "step": 2950 + }, + { + "entropy": 0.19853905692696572, + "epoch": 7.353374233128834, + "grad_norm": 0.8545557260513306, + "learning_rate": 8.032288189962588e-05, + "loss": 0.1555456066131592, + "mean_token_accuracy": 0.9425770407915115, + "num_tokens": 7941760.0, + "step": 3000 + }, + { + "entropy": 0.19454577103257178, + "epoch": 7.476073619631902, + "grad_norm": 0.9338676929473877, + "learning_rate": 7.354302733522059e-05, + "loss": 0.15240780830383302, + "mean_token_accuracy": 0.9427168095111846, + "num_tokens": 8080328.0, + "step": 3050 + }, + { + "entropy": 0.19621057718992232, + "epoch": 7.598773006134969, + "grad_norm": 0.920280933380127, + "learning_rate": 6.699782319135416e-05, + "loss": 0.1560393238067627, + "mean_token_accuracy": 0.9418092548847199, + "num_tokens": 8210831.0, + "step": 3100 + }, + { + "entropy": 0.19532680556178092, + "epoch": 7.721472392638037, + "grad_norm": 0.8807936906814575, + "learning_rate": 6.069924490521774e-05, + "loss": 0.15486297607421876, + "mean_token_accuracy": 0.9418806570768357, + "num_tokens": 8344985.0, + "step": 3150 + }, + { + "entropy": 0.19740462571382522, + "epoch": 7.844171779141105, + "grad_norm": 1.0521959066390991, + "learning_rate": 5.4658816674834886e-05, + "loss": 0.15762215614318847, + "mean_token_accuracy": 0.9403509825468064, + "num_tokens": 8475716.0, + "step": 3200 + }, + { + "entropy": 0.1947207669913769, + "epoch": 7.9668711656441715, + "grad_norm": 0.9047127366065979, + "learning_rate": 4.888759037380488e-05, + "loss": 0.15496024131774902, + "mean_token_accuracy": 0.9418564343452454, + "num_tokens": 8606709.0, + "step": 3250 + }, + { + "epoch": 8.0, + "eval_entropy": 0.34109559910637993, + "eval_mean_token_accuracy": 0.7936825769288199, + "eval_not_syn_loss": 0.9095119833946228, + "eval_not_syn_runtime": 56.9258, + "eval_not_syn_samples_per_second": 24.523, + "eval_not_syn_steps_per_second": 3.074, + "eval_num_tokens": 8641568.0, + "step": 3264 + }, + { + "epoch": 8.0, + "eval_entropy": 0.31804646364280154, + "eval_mean_token_accuracy": 0.8126268771716526, + "eval_num_tokens": 8641568.0, + "eval_syn_loss": 0.8467251658439636, + "eval_syn_runtime": 58.7085, + "eval_syn_samples_per_second": 23.779, + "eval_syn_steps_per_second": 2.981, + "step": 3264 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.7583979889495923e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2d33c455fc458afa0f96ad0ead1e339ef647195a --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-3672/trainer_state.json @@ -0,0 +1,962 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 9.0, + "eval_steps": 500, + "global_step": 3672, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + }, + { + "entropy": 0.4960306563762703, + "epoch": 5.024539877300613, + "grad_norm": 0.9411171078681946, + "learning_rate": 0.0002353013459391488, + "loss": 0.4484004211425781, + "mean_token_accuracy": 0.8438825571175778, + "num_tokens": 5427362.0, + "step": 2050 + }, + { + "entropy": 0.3916636416316032, + "epoch": 5.147239263803681, + "grad_norm": 0.8929405808448792, + "learning_rate": 0.00022676095505345462, + "loss": 0.3446478271484375, + "mean_token_accuracy": 0.8732858788967133, + "num_tokens": 5564186.0, + "step": 2100 + }, + { + "entropy": 0.39636621803045274, + "epoch": 5.269938650306749, + "grad_norm": 0.69569331407547, + "learning_rate": 0.0002181748793031656, + "loss": 0.3548049163818359, + "mean_token_accuracy": 0.8699262255430221, + "num_tokens": 5694728.0, + "step": 2150 + }, + { + "entropy": 0.4028259950876236, + "epoch": 5.392638036809816, + "grad_norm": 0.7605823874473572, + "learning_rate": 0.00020955882820758868, + "loss": 0.35218334197998047, + "mean_token_accuracy": 0.8704028391838073, + "num_tokens": 5832581.0, + "step": 2200 + }, + { + "entropy": 0.4091260200738907, + "epoch": 5.515337423312883, + "grad_norm": 0.8990920782089233, + "learning_rate": 0.00020092856613044126, + "loss": 0.36372703552246094, + "mean_token_accuracy": 0.8663889598846436, + "num_tokens": 5962417.0, + "step": 2250 + }, + { + "entropy": 0.4057882487773895, + "epoch": 5.638036809815951, + "grad_norm": 0.9872731566429138, + "learning_rate": 0.0001922998834365732, + "loss": 0.3588364791870117, + "mean_token_accuracy": 0.8686526268720627, + "num_tokens": 6092338.0, + "step": 2300 + }, + { + "entropy": 0.41128933161497117, + "epoch": 5.7607361963190185, + "grad_norm": 0.8626382350921631, + "learning_rate": 0.0001836885676011143, + "loss": 0.36508998870849607, + "mean_token_accuracy": 0.8664334756135941, + "num_tokens": 6226327.0, + "step": 2350 + }, + { + "entropy": 0.40721403241157533, + "epoch": 5.883435582822086, + "grad_norm": 0.775476336479187, + "learning_rate": 0.00017511037432391027, + "loss": 0.36363380432128906, + "mean_token_accuracy": 0.8656364333629608, + "num_tokens": 6359083.0, + "step": 2400 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4985327124595642, + "eval_mean_token_accuracy": 0.8019610796655927, + "eval_not_syn_loss": 0.717173159122467, + "eval_not_syn_runtime": 56.4169, + "eval_not_syn_samples_per_second": 24.744, + "eval_not_syn_steps_per_second": 3.102, + "eval_num_tokens": 6481176.0, + "step": 2448 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4651233986445836, + "eval_mean_token_accuracy": 0.8059938202585493, + "eval_num_tokens": 6481176.0, + "eval_syn_loss": 0.6778863668441772, + "eval_syn_runtime": 58.5069, + "eval_syn_samples_per_second": 23.86, + "eval_syn_steps_per_second": 2.991, + "step": 2448 + }, + { + "entropy": 0.40582828509687174, + "epoch": 6.004907975460123, + "grad_norm": 0.6788471341133118, + "learning_rate": 0.0001665809987020958, + "loss": 0.3566559600830078, + "mean_token_accuracy": 0.868294438930473, + "num_tokens": 6486549.0, + "step": 2450 + }, + { + "entropy": 0.2933592677116394, + "epoch": 6.12760736196319, + "grad_norm": 0.6598446369171143, + "learning_rate": 0.00015811604651354912, + "loss": 0.24192693710327148, + "mean_token_accuracy": 0.9082661873102188, + "num_tokens": 6622384.0, + "step": 2500 + }, + { + "entropy": 0.2975004023313522, + "epoch": 6.250306748466258, + "grad_norm": 0.7413120865821838, + "learning_rate": 0.00014973100566377046, + "loss": 0.2502699661254883, + "mean_token_accuracy": 0.9056042343378067, + "num_tokens": 6758637.0, + "step": 2550 + }, + { + "entropy": 0.3052474121749401, + "epoch": 6.373006134969325, + "grad_norm": 0.8629281520843506, + "learning_rate": 0.00014144121784842467, + "loss": 0.2543815040588379, + "mean_token_accuracy": 0.9046169209480286, + "num_tokens": 6888533.0, + "step": 2600 + }, + { + "entropy": 0.303673956990242, + "epoch": 6.495705521472392, + "grad_norm": 0.8680911660194397, + "learning_rate": 0.000133261850483398, + "loss": 0.2529561424255371, + "mean_token_accuracy": 0.9040863698720932, + "num_tokens": 7023729.0, + "step": 2650 + }, + { + "entropy": 0.2993183335661888, + "epoch": 6.61840490797546, + "grad_norm": 0.7648728489875793, + "learning_rate": 0.00012520786895372493, + "loss": 0.2517524528503418, + "mean_token_accuracy": 0.9038773095607757, + "num_tokens": 7153821.0, + "step": 2700 + }, + { + "entropy": 0.3006903246045113, + "epoch": 6.741104294478528, + "grad_norm": 0.7433948516845703, + "learning_rate": 0.00011729400923216141, + "loss": 0.25459712982177735, + "mean_token_accuracy": 0.9037652736902237, + "num_tokens": 7283513.0, + "step": 2750 + }, + { + "entropy": 0.30416361182928087, + "epoch": 6.863803680981595, + "grad_norm": 0.8744818568229675, + "learning_rate": 0.00010953475091750244, + "loss": 0.25621614456176756, + "mean_token_accuracy": 0.9019701254367828, + "num_tokens": 7411578.0, + "step": 2800 + }, + { + "entropy": 0.2956543755531311, + "epoch": 6.986503067484662, + "grad_norm": 0.8242411017417908, + "learning_rate": 0.00010194429074197415, + "loss": 0.24829059600830078, + "mean_token_accuracy": 0.9054388856887817, + "num_tokens": 7548356.0, + "step": 2850 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4070556444781167, + "eval_mean_token_accuracy": 0.7848251972879682, + "eval_not_syn_loss": 0.8054981827735901, + "eval_not_syn_runtime": 56.6822, + "eval_not_syn_samples_per_second": 24.629, + "eval_not_syn_steps_per_second": 3.087, + "eval_num_tokens": 7561372.0, + "step": 2856 + }, + { + "epoch": 7.0, + "eval_entropy": 0.3784746325016022, + "eval_mean_token_accuracy": 0.8244705755370004, + "eval_num_tokens": 7561372.0, + "eval_syn_loss": 0.7377892732620239, + "eval_syn_runtime": 58.6534, + "eval_syn_samples_per_second": 23.801, + "eval_syn_steps_per_second": 2.984, + "step": 2856 + }, + { + "entropy": 0.22089176542229122, + "epoch": 7.1079754601227, + "grad_norm": 0.7850068211555481, + "learning_rate": 9.453651659617315e-05, + "loss": 0.17431402206420898, + "mean_token_accuracy": 0.9356574006754943, + "num_tokens": 7671657.0, + "step": 2900 + }, + { + "entropy": 0.2014119729399681, + "epoch": 7.230674846625767, + "grad_norm": 0.6396872401237488, + "learning_rate": 8.732498211907838e-05, + "loss": 0.15425819396972656, + "mean_token_accuracy": 0.9421506607532502, + "num_tokens": 7806989.0, + "step": 2950 + }, + { + "entropy": 0.19853905692696572, + "epoch": 7.353374233128834, + "grad_norm": 0.8545557260513306, + "learning_rate": 8.032288189962588e-05, + "loss": 0.1555456066131592, + "mean_token_accuracy": 0.9425770407915115, + "num_tokens": 7941760.0, + "step": 3000 + }, + { + "entropy": 0.19454577103257178, + "epoch": 7.476073619631902, + "grad_norm": 0.9338676929473877, + "learning_rate": 7.354302733522059e-05, + "loss": 0.15240780830383302, + "mean_token_accuracy": 0.9427168095111846, + "num_tokens": 8080328.0, + "step": 3050 + }, + { + "entropy": 0.19621057718992232, + "epoch": 7.598773006134969, + "grad_norm": 0.920280933380127, + "learning_rate": 6.699782319135416e-05, + "loss": 0.1560393238067627, + "mean_token_accuracy": 0.9418092548847199, + "num_tokens": 8210831.0, + "step": 3100 + }, + { + "entropy": 0.19532680556178092, + "epoch": 7.721472392638037, + "grad_norm": 0.8807936906814575, + "learning_rate": 6.069924490521774e-05, + "loss": 0.15486297607421876, + "mean_token_accuracy": 0.9418806570768357, + "num_tokens": 8344985.0, + "step": 3150 + }, + { + "entropy": 0.19740462571382522, + "epoch": 7.844171779141105, + "grad_norm": 1.0521959066390991, + "learning_rate": 5.4658816674834886e-05, + "loss": 0.15762215614318847, + "mean_token_accuracy": 0.9403509825468064, + "num_tokens": 8475716.0, + "step": 3200 + }, + { + "entropy": 0.1947207669913769, + "epoch": 7.9668711656441715, + "grad_norm": 0.9047127366065979, + "learning_rate": 4.888759037380488e-05, + "loss": 0.15496024131774902, + "mean_token_accuracy": 0.9418564343452454, + "num_tokens": 8606709.0, + "step": 3250 + }, + { + "epoch": 8.0, + "eval_entropy": 0.34109559910637993, + "eval_mean_token_accuracy": 0.7936825769288199, + "eval_not_syn_loss": 0.9095119833946228, + "eval_not_syn_runtime": 56.9258, + "eval_not_syn_samples_per_second": 24.523, + "eval_not_syn_steps_per_second": 3.074, + "eval_num_tokens": 8641568.0, + "step": 3264 + }, + { + "epoch": 8.0, + "eval_entropy": 0.31804646364280154, + "eval_mean_token_accuracy": 0.8126268771716526, + "eval_num_tokens": 8641568.0, + "eval_syn_loss": 0.8467251658439636, + "eval_syn_runtime": 58.7085, + "eval_syn_samples_per_second": 23.779, + "eval_syn_steps_per_second": 2.981, + "step": 3264 + }, + { + "entropy": 0.16006369128672762, + "epoch": 8.088343558282208, + "grad_norm": 0.6774137616157532, + "learning_rate": 4.339612533023478e-05, + "loss": 0.11336767196655273, + "mean_token_accuracy": 0.9597395754823781, + "num_tokens": 8733256.0, + "step": 3300 + }, + { + "entropy": 0.13934748992323875, + "epoch": 8.211042944785277, + "grad_norm": 0.7012852430343628, + "learning_rate": 3.8194469006857826e-05, + "loss": 0.09432272911071778, + "mean_token_accuracy": 0.9670496737957001, + "num_tokens": 8864507.0, + "step": 3350 + }, + { + "entropy": 0.13847702488303185, + "epoch": 8.333742331288343, + "grad_norm": 0.6204716563224792, + "learning_rate": 3.329213861768602e-05, + "loss": 0.09572757720947266, + "mean_token_accuracy": 0.9653251791000366, + "num_tokens": 8993933.0, + "step": 3400 + }, + { + "entropy": 0.1341943299770355, + "epoch": 8.45644171779141, + "grad_norm": 0.6529184579849243, + "learning_rate": 2.8698103714833297e-05, + "loss": 0.09508653640747071, + "mean_token_accuracy": 0.9664993113279343, + "num_tokens": 9124819.0, + "step": 3450 + }, + { + "entropy": 0.13126649804413318, + "epoch": 8.579141104294479, + "grad_norm": 0.6677636504173279, + "learning_rate": 2.4420769777368618e-05, + "loss": 0.09260921478271485, + "mean_token_accuracy": 0.967041677236557, + "num_tokens": 9266865.0, + "step": 3500 + }, + { + "entropy": 0.13248525604605674, + "epoch": 8.701840490797546, + "grad_norm": 0.7775622010231018, + "learning_rate": 2.0467962832225135e-05, + "loss": 0.09405729293823242, + "mean_token_accuracy": 0.9667006832361221, + "num_tokens": 9402291.0, + "step": 3550 + }, + { + "entropy": 0.13282355941832066, + "epoch": 8.824539877300614, + "grad_norm": 0.6643354296684265, + "learning_rate": 1.6846915135304847e-05, + "loss": 0.09236414909362793, + "mean_token_accuracy": 0.967280547618866, + "num_tokens": 9536676.0, + "step": 3600 + }, + { + "entropy": 0.13367731645703315, + "epoch": 8.94723926380368, + "grad_norm": 0.7113834023475647, + "learning_rate": 1.3564251938976921e-05, + "loss": 0.09302605628967285, + "mean_token_accuracy": 0.9672682428359985, + "num_tokens": 9668941.0, + "step": 3650 + }, + { + "epoch": 9.0, + "eval_entropy": 0.30514796699796404, + "eval_mean_token_accuracy": 0.7953711404119219, + "eval_not_syn_loss": 1.0386866331100464, + "eval_not_syn_runtime": 56.9221, + "eval_not_syn_samples_per_second": 24.525, + "eval_not_syn_steps_per_second": 3.074, + "eval_num_tokens": 9721764.0, + "step": 3672 + }, + { + "epoch": 9.0, + "eval_entropy": 0.28337277259145466, + "eval_mean_token_accuracy": 0.8083208533695766, + "eval_num_tokens": 9721764.0, + "eval_syn_loss": 0.9751896262168884, + "eval_syn_runtime": 58.4388, + "eval_syn_samples_per_second": 23.888, + "eval_syn_steps_per_second": 2.995, + "step": 3672 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.9782483744860506e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6ee0790c390bc635393f04c95116bbc5614f6737 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-408/trainer_state.json @@ -0,0 +1,136 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 408, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.2127480787778944e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..47e42076eba8e4d7fd1caf467481447eb4663dbe --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-4080/trainer_state.json @@ -0,0 +1,1064 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 10.0, + "eval_steps": 500, + "global_step": 4080, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + }, + { + "entropy": 0.6619202801675508, + "epoch": 2.083435582822086, + "grad_norm": 0.8124546408653259, + "learning_rate": 0.00038939014724212456, + "loss": 0.6311377716064454, + "mean_token_accuracy": 0.7991988044796567, + "num_tokens": 2253179.0, + "step": 850 + }, + { + "entropy": 0.6668815296888352, + "epoch": 2.2061349693251535, + "grad_norm": 0.8693709969520569, + "learning_rate": 0.0003860393066780265, + "loss": 0.6279015731811524, + "mean_token_accuracy": 0.7983763587474823, + "num_tokens": 2379965.0, + "step": 900 + }, + { + "entropy": 0.6557431703805924, + "epoch": 2.3288343558282207, + "grad_norm": 0.8323768377304077, + "learning_rate": 0.0003823513575054986, + "loss": 0.6214478683471679, + "mean_token_accuracy": 0.8012632429599762, + "num_tokens": 2515593.0, + "step": 950 + }, + { + "entropy": 0.6633540654182434, + "epoch": 2.4515337423312884, + "grad_norm": 0.8328827023506165, + "learning_rate": 0.0003783330473832399, + "loss": 0.6310849761962891, + "mean_token_accuracy": 0.7995503985881806, + "num_tokens": 2642790.0, + "step": 1000 + }, + { + "entropy": 0.6542247200012207, + "epoch": 2.574233128834356, + "grad_norm": 0.8204748630523682, + "learning_rate": 0.000373991728415085, + "loss": 0.6195787048339844, + "mean_token_accuracy": 0.8006457507610321, + "num_tokens": 2774396.0, + "step": 1050 + }, + { + "entropy": 0.6526653742790223, + "epoch": 2.6969325153374233, + "grad_norm": 1.065635323524475, + "learning_rate": 0.0003693353436982218, + "loss": 0.6206568145751953, + "mean_token_accuracy": 0.8002620726823807, + "num_tokens": 2912102.0, + "step": 1100 + }, + { + "entropy": 0.654406590461731, + "epoch": 2.819631901840491, + "grad_norm": 0.8653210401535034, + "learning_rate": 0.00036437241279009834, + "loss": 0.6259954452514649, + "mean_token_accuracy": 0.8008036535978317, + "num_tokens": 3042987.0, + "step": 1150 + }, + { + "entropy": 0.6310120761394501, + "epoch": 2.942331288343558, + "grad_norm": 0.7713245749473572, + "learning_rate": 0.0003591120161206093, + "loss": 0.6115898132324219, + "mean_token_accuracy": 0.8043569129705429, + "num_tokens": 3178044.0, + "step": 1200 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6591688336644854, + "eval_mean_token_accuracy": 0.7992405581474304, + "eval_not_syn_loss": 0.6962469220161438, + "eval_not_syn_runtime": 56.161, + "eval_not_syn_samples_per_second": 24.857, + "eval_not_syn_steps_per_second": 3.116, + "eval_num_tokens": 3240588.0, + "step": 1224 + }, + { + "epoch": 3.0, + "eval_entropy": 0.6191406740461077, + "eval_mean_token_accuracy": 0.7810143651281084, + "eval_num_tokens": 3240588.0, + "eval_syn_loss": 0.6663697361946106, + "eval_syn_runtime": 58.4102, + "eval_syn_samples_per_second": 23.9, + "eval_syn_steps_per_second": 2.996, + "step": 1224 + }, + { + "entropy": 0.5853322166385073, + "epoch": 3.063803680981595, + "grad_norm": 0.7313649654388428, + "learning_rate": 0.00035356377837808084, + "loss": 0.5616408538818359, + "mean_token_accuracy": 0.8144418639366073, + "num_tokens": 3312797.0, + "step": 1250 + }, + { + "entropy": 0.5654504173994064, + "epoch": 3.1865030674846624, + "grad_norm": 0.713893711566925, + "learning_rate": 0.0003477378508994529, + "loss": 0.5305458068847656, + "mean_token_accuracy": 0.8218590825796127, + "num_tokens": 3441629.0, + "step": 1300 + }, + { + "entropy": 0.5721602493524551, + "epoch": 3.30920245398773, + "grad_norm": 0.8487221002578735, + "learning_rate": 0.00034164489309687927, + "loss": 0.5437272262573242, + "mean_token_accuracy": 0.8169907212257386, + "num_tokens": 3576093.0, + "step": 1350 + }, + { + "entropy": 0.5797226822376251, + "epoch": 3.4319018404907977, + "grad_norm": 0.9307994842529297, + "learning_rate": 0.0003352960529547267, + "loss": 0.5405152130126953, + "mean_token_accuracy": 0.818563597202301, + "num_tokens": 3705912.0, + "step": 1400 + }, + { + "entropy": 0.5822264245152473, + "epoch": 3.554601226993865, + "grad_norm": 0.71127849817276, + "learning_rate": 0.00032870294663265757, + "loss": 0.5559784698486329, + "mean_token_accuracy": 0.8159428876638413, + "num_tokens": 3832376.0, + "step": 1450 + }, + { + "entropy": 0.5702577340602875, + "epoch": 3.6773006134969326, + "grad_norm": 0.8090758919715881, + "learning_rate": 0.0003218776372121157, + "loss": 0.5466982650756836, + "mean_token_accuracy": 0.8186432421207428, + "num_tokens": 3967273.0, + "step": 1500 + }, + { + "entropy": 0.5708746567368508, + "epoch": 3.8, + "grad_norm": 0.7777291536331177, + "learning_rate": 0.000314832612625101, + "loss": 0.5412663650512696, + "mean_token_accuracy": 0.8195344746112824, + "num_tokens": 4102434.0, + "step": 1550 + }, + { + "entropy": 0.578739655315876, + "epoch": 3.9226993865030675, + "grad_norm": 0.8103565573692322, + "learning_rate": 0.0003075807628056167, + "loss": 0.5605202865600586, + "mean_token_accuracy": 0.815134603381157, + "num_tokens": 4234037.0, + "step": 1600 + }, + { + "epoch": 4.0, + "eval_entropy": 0.6183825404303415, + "eval_mean_token_accuracy": 0.7838981601170131, + "eval_not_syn_loss": 0.6814553737640381, + "eval_not_syn_runtime": 56.4858, + "eval_not_syn_samples_per_second": 24.714, + "eval_not_syn_steps_per_second": 3.098, + "eval_num_tokens": 4320784.0, + "step": 1632 + }, + { + "epoch": 4.0, + "eval_entropy": 0.5816301565510886, + "eval_mean_token_accuracy": 0.8187636685371399, + "eval_num_tokens": 4320784.0, + "eval_syn_loss": 0.6366032361984253, + "eval_syn_runtime": 58.1317, + "eval_syn_samples_per_second": 24.014, + "eval_syn_steps_per_second": 3.01, + "step": 1632 + }, + { + "entropy": 0.5312421954039371, + "epoch": 4.044171779141104, + "grad_norm": 0.6530773043632507, + "learning_rate": 0.00030013535610559226, + "loss": 0.5023544311523438, + "mean_token_accuracy": 0.8297024351177793, + "num_tokens": 4368268.0, + "step": 1650 + }, + { + "entropy": 0.4853435277938843, + "epoch": 4.166871165644172, + "grad_norm": 1.0383695363998413, + "learning_rate": 0.00029251001501843485, + "loss": 0.45016471862792967, + "mean_token_accuracy": 0.8418685424327851, + "num_tokens": 4494525.0, + "step": 1700 + }, + { + "entropy": 0.48153585255146025, + "epoch": 4.289570552147239, + "grad_norm": 0.7215332388877869, + "learning_rate": 0.00028471869125462477, + "loss": 0.4460752487182617, + "mean_token_accuracy": 0.8421206694841384, + "num_tokens": 4631985.0, + "step": 1750 + }, + { + "entropy": 0.49278041124343874, + "epoch": 4.412269938650307, + "grad_norm": 0.7771472334861755, + "learning_rate": 0.0002767756402149588, + "loss": 0.45794551849365234, + "mean_token_accuracy": 0.8404830145835877, + "num_tokens": 4766323.0, + "step": 1800 + }, + { + "entropy": 0.5082326257228851, + "epoch": 4.534969325153375, + "grad_norm": 0.9560676217079163, + "learning_rate": 0.00026869539490814704, + "loss": 0.4632451629638672, + "mean_token_accuracy": 0.8386451864242553, + "num_tokens": 4895606.0, + "step": 1850 + }, + { + "entropy": 0.5081156292557716, + "epoch": 4.6576687116564415, + "grad_norm": 0.7719491720199585, + "learning_rate": 0.0002604927393604828, + "loss": 0.4630893325805664, + "mean_token_accuracy": 0.839306333065033, + "num_tokens": 5025912.0, + "step": 1900 + }, + { + "entropy": 0.5121548187732696, + "epoch": 4.780368098159509, + "grad_norm": 0.8171904683113098, + "learning_rate": 0.00025218268156623985, + "loss": 0.46541053771972657, + "mean_token_accuracy": 0.8369353520870209, + "num_tokens": 5162972.0, + "step": 1950 + }, + { + "entropy": 0.5085560208559037, + "epoch": 4.903067484662577, + "grad_norm": 0.5919133424758911, + "learning_rate": 0.00024378042602828492, + "loss": 0.46405506134033203, + "mean_token_accuracy": 0.8379521882534027, + "num_tokens": 5297935.0, + "step": 2000 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5431888212476458, + "eval_mean_token_accuracy": 0.7749751152311053, + "eval_not_syn_loss": 0.7035888433456421, + "eval_not_syn_runtime": 56.5207, + "eval_not_syn_samples_per_second": 24.699, + "eval_not_syn_steps_per_second": 3.096, + "eval_num_tokens": 5400980.0, + "step": 2040 + }, + { + "epoch": 5.0, + "eval_entropy": 0.5098190077713558, + "eval_mean_token_accuracy": 0.832349077633449, + "eval_num_tokens": 5400980.0, + "eval_syn_loss": 0.6227898597717285, + "eval_syn_runtime": 57.6978, + "eval_syn_samples_per_second": 24.195, + "eval_syn_steps_per_second": 3.033, + "step": 2040 + }, + { + "entropy": 0.4960306563762703, + "epoch": 5.024539877300613, + "grad_norm": 0.9411171078681946, + "learning_rate": 0.0002353013459391488, + "loss": 0.4484004211425781, + "mean_token_accuracy": 0.8438825571175778, + "num_tokens": 5427362.0, + "step": 2050 + }, + { + "entropy": 0.3916636416316032, + "epoch": 5.147239263803681, + "grad_norm": 0.8929405808448792, + "learning_rate": 0.00022676095505345462, + "loss": 0.3446478271484375, + "mean_token_accuracy": 0.8732858788967133, + "num_tokens": 5564186.0, + "step": 2100 + }, + { + "entropy": 0.39636621803045274, + "epoch": 5.269938650306749, + "grad_norm": 0.69569331407547, + "learning_rate": 0.0002181748793031656, + "loss": 0.3548049163818359, + "mean_token_accuracy": 0.8699262255430221, + "num_tokens": 5694728.0, + "step": 2150 + }, + { + "entropy": 0.4028259950876236, + "epoch": 5.392638036809816, + "grad_norm": 0.7605823874473572, + "learning_rate": 0.00020955882820758868, + "loss": 0.35218334197998047, + "mean_token_accuracy": 0.8704028391838073, + "num_tokens": 5832581.0, + "step": 2200 + }, + { + "entropy": 0.4091260200738907, + "epoch": 5.515337423312883, + "grad_norm": 0.8990920782089233, + "learning_rate": 0.00020092856613044126, + "loss": 0.36372703552246094, + "mean_token_accuracy": 0.8663889598846436, + "num_tokens": 5962417.0, + "step": 2250 + }, + { + "entropy": 0.4057882487773895, + "epoch": 5.638036809815951, + "grad_norm": 0.9872731566429138, + "learning_rate": 0.0001922998834365732, + "loss": 0.3588364791870117, + "mean_token_accuracy": 0.8686526268720627, + "num_tokens": 6092338.0, + "step": 2300 + }, + { + "entropy": 0.41128933161497117, + "epoch": 5.7607361963190185, + "grad_norm": 0.8626382350921631, + "learning_rate": 0.0001836885676011143, + "loss": 0.36508998870849607, + "mean_token_accuracy": 0.8664334756135941, + "num_tokens": 6226327.0, + "step": 2350 + }, + { + "entropy": 0.40721403241157533, + "epoch": 5.883435582822086, + "grad_norm": 0.775476336479187, + "learning_rate": 0.00017511037432391027, + "loss": 0.36363380432128906, + "mean_token_accuracy": 0.8656364333629608, + "num_tokens": 6359083.0, + "step": 2400 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4985327124595642, + "eval_mean_token_accuracy": 0.8019610796655927, + "eval_not_syn_loss": 0.717173159122467, + "eval_not_syn_runtime": 56.4169, + "eval_not_syn_samples_per_second": 24.744, + "eval_not_syn_steps_per_second": 3.102, + "eval_num_tokens": 6481176.0, + "step": 2448 + }, + { + "epoch": 6.0, + "eval_entropy": 0.4651233986445836, + "eval_mean_token_accuracy": 0.8059938202585493, + "eval_num_tokens": 6481176.0, + "eval_syn_loss": 0.6778863668441772, + "eval_syn_runtime": 58.5069, + "eval_syn_samples_per_second": 23.86, + "eval_syn_steps_per_second": 2.991, + "step": 2448 + }, + { + "entropy": 0.40582828509687174, + "epoch": 6.004907975460123, + "grad_norm": 0.6788471341133118, + "learning_rate": 0.0001665809987020958, + "loss": 0.3566559600830078, + "mean_token_accuracy": 0.868294438930473, + "num_tokens": 6486549.0, + "step": 2450 + }, + { + "entropy": 0.2933592677116394, + "epoch": 6.12760736196319, + "grad_norm": 0.6598446369171143, + "learning_rate": 0.00015811604651354912, + "loss": 0.24192693710327148, + "mean_token_accuracy": 0.9082661873102188, + "num_tokens": 6622384.0, + "step": 2500 + }, + { + "entropy": 0.2975004023313522, + "epoch": 6.250306748466258, + "grad_norm": 0.7413120865821838, + "learning_rate": 0.00014973100566377046, + "loss": 0.2502699661254883, + "mean_token_accuracy": 0.9056042343378067, + "num_tokens": 6758637.0, + "step": 2550 + }, + { + "entropy": 0.3052474121749401, + "epoch": 6.373006134969325, + "grad_norm": 0.8629281520843506, + "learning_rate": 0.00014144121784842467, + "loss": 0.2543815040588379, + "mean_token_accuracy": 0.9046169209480286, + "num_tokens": 6888533.0, + "step": 2600 + }, + { + "entropy": 0.303673956990242, + "epoch": 6.495705521472392, + "grad_norm": 0.8680911660194397, + "learning_rate": 0.000133261850483398, + "loss": 0.2529561424255371, + "mean_token_accuracy": 0.9040863698720932, + "num_tokens": 7023729.0, + "step": 2650 + }, + { + "entropy": 0.2993183335661888, + "epoch": 6.61840490797546, + "grad_norm": 0.7648728489875793, + "learning_rate": 0.00012520786895372493, + "loss": 0.2517524528503418, + "mean_token_accuracy": 0.9038773095607757, + "num_tokens": 7153821.0, + "step": 2700 + }, + { + "entropy": 0.3006903246045113, + "epoch": 6.741104294478528, + "grad_norm": 0.7433948516845703, + "learning_rate": 0.00011729400923216141, + "loss": 0.25459712982177735, + "mean_token_accuracy": 0.9037652736902237, + "num_tokens": 7283513.0, + "step": 2750 + }, + { + "entropy": 0.30416361182928087, + "epoch": 6.863803680981595, + "grad_norm": 0.8744818568229675, + "learning_rate": 0.00010953475091750244, + "loss": 0.25621614456176756, + "mean_token_accuracy": 0.9019701254367828, + "num_tokens": 7411578.0, + "step": 2800 + }, + { + "entropy": 0.2956543755531311, + "epoch": 6.986503067484662, + "grad_norm": 0.8242411017417908, + "learning_rate": 0.00010194429074197415, + "loss": 0.24829059600830078, + "mean_token_accuracy": 0.9054388856887817, + "num_tokens": 7548356.0, + "step": 2850 + }, + { + "epoch": 7.0, + "eval_entropy": 0.4070556444781167, + "eval_mean_token_accuracy": 0.7848251972879682, + "eval_not_syn_loss": 0.8054981827735901, + "eval_not_syn_runtime": 56.6822, + "eval_not_syn_samples_per_second": 24.629, + "eval_not_syn_steps_per_second": 3.087, + "eval_num_tokens": 7561372.0, + "step": 2856 + }, + { + "epoch": 7.0, + "eval_entropy": 0.3784746325016022, + "eval_mean_token_accuracy": 0.8244705755370004, + "eval_num_tokens": 7561372.0, + "eval_syn_loss": 0.7377892732620239, + "eval_syn_runtime": 58.6534, + "eval_syn_samples_per_second": 23.801, + "eval_syn_steps_per_second": 2.984, + "step": 2856 + }, + { + "entropy": 0.22089176542229122, + "epoch": 7.1079754601227, + "grad_norm": 0.7850068211555481, + "learning_rate": 9.453651659617315e-05, + "loss": 0.17431402206420898, + "mean_token_accuracy": 0.9356574006754943, + "num_tokens": 7671657.0, + "step": 2900 + }, + { + "entropy": 0.2014119729399681, + "epoch": 7.230674846625767, + "grad_norm": 0.6396872401237488, + "learning_rate": 8.732498211907838e-05, + "loss": 0.15425819396972656, + "mean_token_accuracy": 0.9421506607532502, + "num_tokens": 7806989.0, + "step": 2950 + }, + { + "entropy": 0.19853905692696572, + "epoch": 7.353374233128834, + "grad_norm": 0.8545557260513306, + "learning_rate": 8.032288189962588e-05, + "loss": 0.1555456066131592, + "mean_token_accuracy": 0.9425770407915115, + "num_tokens": 7941760.0, + "step": 3000 + }, + { + "entropy": 0.19454577103257178, + "epoch": 7.476073619631902, + "grad_norm": 0.9338676929473877, + "learning_rate": 7.354302733522059e-05, + "loss": 0.15240780830383302, + "mean_token_accuracy": 0.9427168095111846, + "num_tokens": 8080328.0, + "step": 3050 + }, + { + "entropy": 0.19621057718992232, + "epoch": 7.598773006134969, + "grad_norm": 0.920280933380127, + "learning_rate": 6.699782319135416e-05, + "loss": 0.1560393238067627, + "mean_token_accuracy": 0.9418092548847199, + "num_tokens": 8210831.0, + "step": 3100 + }, + { + "entropy": 0.19532680556178092, + "epoch": 7.721472392638037, + "grad_norm": 0.8807936906814575, + "learning_rate": 6.069924490521774e-05, + "loss": 0.15486297607421876, + "mean_token_accuracy": 0.9418806570768357, + "num_tokens": 8344985.0, + "step": 3150 + }, + { + "entropy": 0.19740462571382522, + "epoch": 7.844171779141105, + "grad_norm": 1.0521959066390991, + "learning_rate": 5.4658816674834886e-05, + "loss": 0.15762215614318847, + "mean_token_accuracy": 0.9403509825468064, + "num_tokens": 8475716.0, + "step": 3200 + }, + { + "entropy": 0.1947207669913769, + "epoch": 7.9668711656441715, + "grad_norm": 0.9047127366065979, + "learning_rate": 4.888759037380488e-05, + "loss": 0.15496024131774902, + "mean_token_accuracy": 0.9418564343452454, + "num_tokens": 8606709.0, + "step": 3250 + }, + { + "epoch": 8.0, + "eval_entropy": 0.34109559910637993, + "eval_mean_token_accuracy": 0.7936825769288199, + "eval_not_syn_loss": 0.9095119833946228, + "eval_not_syn_runtime": 56.9258, + "eval_not_syn_samples_per_second": 24.523, + "eval_not_syn_steps_per_second": 3.074, + "eval_num_tokens": 8641568.0, + "step": 3264 + }, + { + "epoch": 8.0, + "eval_entropy": 0.31804646364280154, + "eval_mean_token_accuracy": 0.8126268771716526, + "eval_num_tokens": 8641568.0, + "eval_syn_loss": 0.8467251658439636, + "eval_syn_runtime": 58.7085, + "eval_syn_samples_per_second": 23.779, + "eval_syn_steps_per_second": 2.981, + "step": 3264 + }, + { + "entropy": 0.16006369128672762, + "epoch": 8.088343558282208, + "grad_norm": 0.6774137616157532, + "learning_rate": 4.339612533023478e-05, + "loss": 0.11336767196655273, + "mean_token_accuracy": 0.9597395754823781, + "num_tokens": 8733256.0, + "step": 3300 + }, + { + "entropy": 0.13934748992323875, + "epoch": 8.211042944785277, + "grad_norm": 0.7012852430343628, + "learning_rate": 3.8194469006857826e-05, + "loss": 0.09432272911071778, + "mean_token_accuracy": 0.9670496737957001, + "num_tokens": 8864507.0, + "step": 3350 + }, + { + "entropy": 0.13847702488303185, + "epoch": 8.333742331288343, + "grad_norm": 0.6204716563224792, + "learning_rate": 3.329213861768602e-05, + "loss": 0.09572757720947266, + "mean_token_accuracy": 0.9653251791000366, + "num_tokens": 8993933.0, + "step": 3400 + }, + { + "entropy": 0.1341943299770355, + "epoch": 8.45644171779141, + "grad_norm": 0.6529184579849243, + "learning_rate": 2.8698103714833297e-05, + "loss": 0.09508653640747071, + "mean_token_accuracy": 0.9664993113279343, + "num_tokens": 9124819.0, + "step": 3450 + }, + { + "entropy": 0.13126649804413318, + "epoch": 8.579141104294479, + "grad_norm": 0.6677636504173279, + "learning_rate": 2.4420769777368618e-05, + "loss": 0.09260921478271485, + "mean_token_accuracy": 0.967041677236557, + "num_tokens": 9266865.0, + "step": 3500 + }, + { + "entropy": 0.13248525604605674, + "epoch": 8.701840490797546, + "grad_norm": 0.7775622010231018, + "learning_rate": 2.0467962832225135e-05, + "loss": 0.09405729293823242, + "mean_token_accuracy": 0.9667006832361221, + "num_tokens": 9402291.0, + "step": 3550 + }, + { + "entropy": 0.13282355941832066, + "epoch": 8.824539877300614, + "grad_norm": 0.6643354296684265, + "learning_rate": 1.6846915135304847e-05, + "loss": 0.09236414909362793, + "mean_token_accuracy": 0.967280547618866, + "num_tokens": 9536676.0, + "step": 3600 + }, + { + "entropy": 0.13367731645703315, + "epoch": 8.94723926380368, + "grad_norm": 0.7113834023475647, + "learning_rate": 1.3564251938976921e-05, + "loss": 0.09302605628967285, + "mean_token_accuracy": 0.9672682428359985, + "num_tokens": 9668941.0, + "step": 3650 + }, + { + "epoch": 9.0, + "eval_entropy": 0.30514796699796404, + "eval_mean_token_accuracy": 0.7953711404119219, + "eval_not_syn_loss": 1.0386866331100464, + "eval_not_syn_runtime": 56.9221, + "eval_not_syn_samples_per_second": 24.525, + "eval_not_syn_steps_per_second": 3.074, + "eval_num_tokens": 9721764.0, + "step": 3672 + }, + { + "epoch": 9.0, + "eval_entropy": 0.28337277259145466, + "eval_mean_token_accuracy": 0.8083208533695766, + "eval_num_tokens": 9721764.0, + "eval_syn_loss": 0.9751896262168884, + "eval_syn_runtime": 58.4388, + "eval_syn_samples_per_second": 23.888, + "eval_syn_steps_per_second": 2.995, + "step": 3672 + }, + { + "entropy": 0.12280672442431402, + "epoch": 9.068711656441717, + "grad_norm": 0.5349636673927307, + "learning_rate": 1.062597937017992e-05, + "loss": 0.08112168312072754, + "mean_token_accuracy": 0.9721009038915538, + "num_tokens": 9799022.0, + "step": 3700 + }, + { + "entropy": 0.11758630983531475, + "epoch": 9.191411042944786, + "grad_norm": 0.5184180736541748, + "learning_rate": 8.03747344130768e-06, + "loss": 0.07424304962158203, + "mean_token_accuracy": 0.9749062228202819, + "num_tokens": 9927323.0, + "step": 3750 + }, + { + "entropy": 0.11504819758236408, + "epoch": 9.314110429447853, + "grad_norm": 0.4414903223514557, + "learning_rate": 5.803470213984592e-06, + "loss": 0.07420271396636963, + "mean_token_accuracy": 0.9745804452896119, + "num_tokens": 10057034.0, + "step": 3800 + }, + { + "entropy": 0.11219210438430309, + "epoch": 9.43680981595092, + "grad_norm": 0.6505016684532166, + "learning_rate": 3.928057133727374e-06, + "loss": 0.07242022514343262, + "mean_token_accuracy": 0.9752460938692092, + "num_tokens": 10190986.0, + "step": 3850 + }, + { + "entropy": 0.1172893676161766, + "epoch": 9.559509202453988, + "grad_norm": 0.4053712785243988, + "learning_rate": 2.4146655513473907e-06, + "loss": 0.07574840545654297, + "mean_token_accuracy": 0.9739107710123062, + "num_tokens": 10318963.0, + "step": 3900 + }, + { + "entropy": 0.11008265137672424, + "epoch": 9.682208588957055, + "grad_norm": 0.4532715082168579, + "learning_rate": 1.2660644447774456e-06, + "loss": 0.0717643404006958, + "mean_token_accuracy": 0.9753157407045364, + "num_tokens": 10453673.0, + "step": 3950 + }, + { + "entropy": 0.11246884763240814, + "epoch": 9.804907975460123, + "grad_norm": 0.46811601519584656, + "learning_rate": 4.843553528094893e-07, + "loss": 0.07214241027832032, + "mean_token_accuracy": 0.9749426013231277, + "num_tokens": 10588217.0, + "step": 4000 + }, + { + "entropy": 0.1111858068406582, + "epoch": 9.92760736196319, + "grad_norm": 0.5065276026725769, + "learning_rate": 7.096853001258902e-08, + "loss": 0.07076848030090332, + "mean_token_accuracy": 0.9755072641372681, + "num_tokens": 10725457.0, + "step": 4050 + }, + { + "epoch": 10.0, + "eval_entropy": 0.29344515562057494, + "eval_mean_token_accuracy": 0.7949213702338083, + "eval_not_syn_loss": 1.1034446954727173, + "eval_not_syn_runtime": 55.6095, + "eval_not_syn_samples_per_second": 25.104, + "eval_not_syn_steps_per_second": 3.147, + "eval_num_tokens": 10801960.0, + "step": 4080 + }, + { + "epoch": 10.0, + "eval_entropy": 0.271583269068173, + "eval_mean_token_accuracy": 0.809326388154711, + "eval_num_tokens": 10801960.0, + "eval_syn_loss": 1.03242027759552, + "eval_syn_runtime": 58.6681, + "eval_syn_samples_per_second": 23.795, + "eval_syn_steps_per_second": 2.983, + "step": 4080 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 2.1980494514604096e+17, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/README.md b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ac50e2706e5b4eae8f28e204601afb241c49e55 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/README.md @@ -0,0 +1,209 @@ +--- +base_model: Qwen/Qwen3.5-2B-Base +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen3.5-2B-Base +- lora +- sft +- transformers +- trl +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/adapter_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..100cd582d71fb39c74fbba0e56e6d3415622454e --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "Qwen/Qwen3.5-2B-Base", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 128, + "lora_bias": false, + "lora_dropout": 0.05635041589438875, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 64, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "q_proj", + "k_proj", + "gate_proj", + "up_proj", + "o_proj", + "down_proj", + "v_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/chat_template.jinja b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/tokenizer_config.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/trainer_state.json b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..aafb56dfefab7f93810a88683941f29d1cc3d0e6 --- /dev/null +++ b/substitutivity_original_Estonian/Qwen3.5-2B-Base_substitutivity_splits_original_features_train_substitutivity_splits_original_features_test2/checkpoint-816/trainer_state.json @@ -0,0 +1,238 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 816, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 1.8153233402967452, + "epoch": 0.12269938650306748, + "grad_norm": 1.8510568141937256, + "learning_rate": 4.8469594667113335e-05, + "loss": 1.7217323303222656, + "mean_token_accuracy": 0.6236635231971741, + "num_tokens": 134329.0, + "step": 50 + }, + { + "entropy": 0.97831438601017, + "epoch": 0.24539877300613497, + "grad_norm": 1.889868974685669, + "learning_rate": 9.792836473559631e-05, + "loss": 0.9050540924072266, + "mean_token_accuracy": 0.7377463465929032, + "num_tokens": 262407.0, + "step": 100 + }, + { + "entropy": 0.8743131357431412, + "epoch": 0.36809815950920244, + "grad_norm": 1.1766732931137085, + "learning_rate": 0.0001473871348040793, + "loss": 0.8129763031005859, + "mean_token_accuracy": 0.7583417356014251, + "num_tokens": 401136.0, + "step": 150 + }, + { + "entropy": 0.8273966419696808, + "epoch": 0.49079754601226994, + "grad_norm": 1.1311014890670776, + "learning_rate": 0.00019684590487256229, + "loss": 0.7711280822753906, + "mean_token_accuracy": 0.7668228060007095, + "num_tokens": 531647.0, + "step": 200 + }, + { + "entropy": 0.8101438617706299, + "epoch": 0.6134969325153374, + "grad_norm": 0.8905950784683228, + "learning_rate": 0.0002463046749410453, + "loss": 0.7609222412109375, + "mean_token_accuracy": 0.7717174577713013, + "num_tokens": 666127.0, + "step": 250 + }, + { + "entropy": 0.7885055243968964, + "epoch": 0.7361963190184049, + "grad_norm": 1.1944245100021362, + "learning_rate": 0.00029576344500952824, + "loss": 0.7477743530273437, + "mean_token_accuracy": 0.773262197971344, + "num_tokens": 799292.0, + "step": 300 + }, + { + "entropy": 0.7858527797460556, + "epoch": 0.8588957055214724, + "grad_norm": 1.1311068534851074, + "learning_rate": 0.00034522221507801124, + "loss": 0.7411377716064453, + "mean_token_accuracy": 0.7745680212974548, + "num_tokens": 930614.0, + "step": 350 + }, + { + "entropy": 0.7879797518253326, + "epoch": 0.9815950920245399, + "grad_norm": 0.9370436072349548, + "learning_rate": 0.00039468098514649423, + "loss": 0.7507785034179687, + "mean_token_accuracy": 0.7747444450855255, + "num_tokens": 1061654.0, + "step": 400 + }, + { + "epoch": 1.0, + "eval_entropy": 0.8527693561145238, + "eval_mean_token_accuracy": 0.7730878329277039, + "eval_not_syn_loss": 0.781921923160553, + "eval_not_syn_runtime": 56.9157, + "eval_not_syn_samples_per_second": 24.527, + "eval_not_syn_steps_per_second": 3.075, + "eval_num_tokens": 1080196.0, + "step": 408 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7952005583899362, + "eval_mean_token_accuracy": 0.7705784467288426, + "eval_num_tokens": 1080196.0, + "eval_syn_loss": 0.7461121082305908, + "eval_syn_runtime": 59.5758, + "eval_syn_samples_per_second": 23.432, + "eval_syn_steps_per_second": 2.937, + "step": 408 + }, + { + "entropy": 0.7564875923021875, + "epoch": 1.1030674846625768, + "grad_norm": 1.1334599256515503, + "learning_rate": 0.00040345942966973637, + "loss": 0.7208955383300781, + "mean_token_accuracy": 0.7788144470465304, + "num_tokens": 1199170.0, + "step": 450 + }, + { + "entropy": 0.7427570706605912, + "epoch": 1.2257668711656442, + "grad_norm": 1.0777021646499634, + "learning_rate": 0.00040297229629224153, + "loss": 0.7049168395996094, + "mean_token_accuracy": 0.7830300116539002, + "num_tokens": 1334298.0, + "step": 500 + }, + { + "entropy": 0.7533353108167649, + "epoch": 1.3484662576687116, + "grad_norm": 1.3439561128616333, + "learning_rate": 0.00040211707285041045, + "loss": 0.7102288818359375, + "mean_token_accuracy": 0.7818449640274048, + "num_tokens": 1463189.0, + "step": 550 + }, + { + "entropy": 0.7666342407464981, + "epoch": 1.471165644171779, + "grad_norm": 1.1329729557037354, + "learning_rate": 0.00040089532410439227, + "loss": 0.7270996856689453, + "mean_token_accuracy": 0.778404277563095, + "num_tokens": 1590310.0, + "step": 600 + }, + { + "entropy": 0.7359934949874878, + "epoch": 1.5938650306748468, + "grad_norm": 1.2227331399917603, + "learning_rate": 0.0003993092854276068, + "loss": 0.6996555328369141, + "mean_token_accuracy": 0.7834958267211914, + "num_tokens": 1721836.0, + "step": 650 + }, + { + "entropy": 0.7278932738304138, + "epoch": 1.716564417177914, + "grad_norm": 0.8753179311752319, + "learning_rate": 0.0003973618587167924, + "loss": 0.694002456665039, + "mean_token_accuracy": 0.7849677371978759, + "num_tokens": 1853507.0, + "step": 700 + }, + { + "entropy": 0.7190143716335297, + "epoch": 1.8392638036809816, + "grad_norm": 0.8524742126464844, + "learning_rate": 0.0003950566070825483, + "loss": 0.6890143585205079, + "mean_token_accuracy": 0.7865350896120071, + "num_tokens": 1988113.0, + "step": 750 + }, + { + "entropy": 0.7072037667036056, + "epoch": 1.961963190184049, + "grad_norm": 0.877348005771637, + "learning_rate": 0.00039239774833008714, + "loss": 0.6795162200927735, + "mean_token_accuracy": 0.7885233855247498, + "num_tokens": 2121314.0, + "step": 800 + }, + { + "epoch": 2.0, + "eval_entropy": 0.818787784235818, + "eval_mean_token_accuracy": 0.7573730857031686, + "eval_not_syn_loss": 0.7486392855644226, + "eval_not_syn_runtime": 56.7426, + "eval_not_syn_samples_per_second": 24.602, + "eval_not_syn_steps_per_second": 3.084, + "eval_num_tokens": 2160392.0, + "step": 816 + }, + { + "epoch": 2.0, + "eval_entropy": 0.7632522354807173, + "eval_mean_token_accuracy": 0.8101163625717163, + "eval_num_tokens": 2160392.0, + "eval_syn_loss": 0.6829464435577393, + "eval_syn_runtime": 58.5255, + "eval_syn_samples_per_second": 23.853, + "eval_syn_steps_per_second": 2.99, + "step": 816 + } + ], + "logging_steps": 50, + "max_steps": 4080, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.410561614273088e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +} diff --git a/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/chat_template.jinja b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..0ef09f214eaa6d9bca297988afc1454b5827b2c7 --- /dev/null +++ b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is true %} + {{- '\n' }} + {%- else %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/tokenizer_config.json b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a37b2a6fd3ab3317cd7bac72855be1a843b2bb --- /dev/null +++ b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/tokenizer_config.json @@ -0,0 +1,31 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "split_special_tokens": false, + "tokenizer_class": "TokenizersBackend", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/trainer_state.json b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..67a56b17ccacd3d45157c7fb2fe74cbd86f232fd --- /dev/null +++ b/systematicity_original_Estonian/Qwen3.5-2B-Base_systematicity_splits_original_features_train_systematicity_splits_original_features_test1/checkpoint-914/trainer_state.json @@ -0,0 +1,236 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 914, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "entropy": 2.074948468208313, + "epoch": 0.10952902519167579, + "grad_norm": 2.3511903285980225, + "learning_rate": 1.2698308676050528e-05, + "loss": 1.9153688049316406, + "mean_token_accuracy": 0.6077924159169197, + "num_tokens": 113087.0, + "step": 50 + }, + { + "entropy": 1.054056368470192, + "epoch": 0.21905805038335158, + "grad_norm": 2.3081915378570557, + "learning_rate": 2.5655766508755147e-05, + "loss": 0.9633240509033203, + "mean_token_accuracy": 0.741833764910698, + "num_tokens": 230392.0, + "step": 100 + }, + { + "entropy": 0.9184817910194397, + "epoch": 0.32858707557502737, + "grad_norm": 2.0965280532836914, + "learning_rate": 3.861322434145977e-05, + "loss": 0.8405924987792969, + "mean_token_accuracy": 0.7688284432888031, + "num_tokens": 341746.0, + "step": 150 + }, + { + "entropy": 0.8618861585855484, + "epoch": 0.43811610076670315, + "grad_norm": 2.2506542205810547, + "learning_rate": 5.157068217416438e-05, + "loss": 0.7841030883789063, + "mean_token_accuracy": 0.7796466761827469, + "num_tokens": 460483.0, + "step": 200 + }, + { + "entropy": 0.819869134426117, + "epoch": 0.547645125958379, + "grad_norm": 1.4058395624160767, + "learning_rate": 6.4528140006869e-05, + "loss": 0.7524736785888672, + "mean_token_accuracy": 0.7900535291433335, + "num_tokens": 570627.0, + "step": 250 + }, + { + "entropy": 0.8075837278366089, + "epoch": 0.6571741511500547, + "grad_norm": 1.5815014839172363, + "learning_rate": 7.748559783957362e-05, + "loss": 0.7388697814941406, + "mean_token_accuracy": 0.7914055424928665, + "num_tokens": 677217.0, + "step": 300 + }, + { + "entropy": 0.7868322026729584, + "epoch": 0.7667031763417306, + "grad_norm": 1.2377994060516357, + "learning_rate": 9.044305567227824e-05, + "loss": 0.7226168060302735, + "mean_token_accuracy": 0.7953901988267899, + "num_tokens": 790185.0, + "step": 350 + }, + { + "entropy": 0.7526325100660324, + "epoch": 0.8762322015334063, + "grad_norm": 1.203137755393982, + "learning_rate": 0.00010340051350498286, + "loss": 0.6991680908203125, + "mean_token_accuracy": 0.7992895567417144, + "num_tokens": 910080.0, + "step": 400 + }, + { + "entropy": 0.7626404595375061, + "epoch": 0.9857612267250822, + "grad_norm": 1.1625452041625977, + "learning_rate": 0.00011635797133768749, + "loss": 0.7055020904541016, + "mean_token_accuracy": 0.7997505795955658, + "num_tokens": 1023473.0, + "step": 450 + }, + { + "epoch": 1.0, + "eval_entropy": 0.7441067624659765, + "eval_loss": 0.7209385633468628, + "eval_mean_token_accuracy": 0.7895652093584575, + "eval_num_tokens": 1038250.0, + "eval_runtime": 45.5076, + "eval_samples_per_second": 22.018, + "eval_steps_per_second": 2.769, + "step": 457 + }, + { + "entropy": 0.720724637460227, + "epoch": 1.0941949616648412, + "grad_norm": 1.2537180185317993, + "learning_rate": 0.00011840069618947894, + "loss": 0.6586117553710937, + "mean_token_accuracy": 0.8081232917429221, + "num_tokens": 1128922.0, + "step": 500 + }, + { + "entropy": 0.7084310132265091, + "epoch": 1.203723986856517, + "grad_norm": 1.2347161769866943, + "learning_rate": 0.00011828501915154198, + "loss": 0.650611572265625, + "mean_token_accuracy": 0.808387525677681, + "num_tokens": 1245635.0, + "step": 550 + }, + { + "entropy": 0.706004958152771, + "epoch": 1.3132530120481927, + "grad_norm": 1.2140169143676758, + "learning_rate": 0.00011808319665683977, + "loss": 0.6442465209960937, + "mean_token_accuracy": 0.8122780376672745, + "num_tokens": 1354885.0, + "step": 600 + }, + { + "entropy": 0.6854291236400605, + "epoch": 1.4227820372398685, + "grad_norm": 1.1137559413909912, + "learning_rate": 0.00011779552303848118, + "loss": 0.6293138885498046, + "mean_token_accuracy": 0.8158071845769882, + "num_tokens": 1472232.0, + "step": 650 + }, + { + "entropy": 0.6885070586204529, + "epoch": 1.5323110624315444, + "grad_norm": 1.1074714660644531, + "learning_rate": 0.00011742241783280468, + "loss": 0.6259001541137695, + "mean_token_accuracy": 0.8164563828706741, + "num_tokens": 1587494.0, + "step": 700 + }, + { + "entropy": 0.6932114177942276, + "epoch": 1.6418400876232202, + "grad_norm": 1.1099495887756348, + "learning_rate": 0.00011696442516753647, + "loss": 0.6335408401489258, + "mean_token_accuracy": 0.8156683903932571, + "num_tokens": 1696927.0, + "step": 750 + }, + { + "entropy": 0.6787053138017655, + "epoch": 1.751369112814896, + "grad_norm": 1.2505418062210083, + "learning_rate": 0.00011642221296824764, + "loss": 0.6210909271240235, + "mean_token_accuracy": 0.8153067737817764, + "num_tokens": 1813415.0, + "step": 800 + }, + { + "entropy": 0.6693948286771775, + "epoch": 1.8608981380065717, + "grad_norm": 1.0963624715805054, + "learning_rate": 0.00011579657198426736, + "loss": 0.6105814361572266, + "mean_token_accuracy": 0.819609425663948, + "num_tokens": 1924911.0, + "step": 850 + }, + { + "entropy": 0.6634757363796234, + "epoch": 1.9704271631982475, + "grad_norm": 0.8766036033630371, + "learning_rate": 0.00011508841463547313, + "loss": 0.6100498580932617, + "mean_token_accuracy": 0.8197187691926956, + "num_tokens": 2044650.0, + "step": 900 + }, + { + "epoch": 2.0, + "eval_entropy": 0.6342300275961558, + "eval_loss": 0.6517631411552429, + "eval_mean_token_accuracy": 0.807878240233376, + "eval_num_tokens": 2076500.0, + "eval_runtime": 44.6225, + "eval_samples_per_second": 22.455, + "eval_steps_per_second": 2.824, + "step": 914 + } + ], + "logging_steps": 50, + "max_steps": 4570, + "num_input_tokens_seen": 0, + "num_train_epochs": 10, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.486683349186176e+16, + "train_batch_size": 8, + "trial_name": null, + "trial_params": null +}