graphcodebert-code-clone-detection / experiment_record.json
thealper2's picture
Add GraphCodeBERT clone-detection model
2ef4ea4 verified
Raw
History Blame Contribute Delete
8.6 kB
{
"model_name": "microsoft/graphcodebert-base",
"model_dir": "/mnt/d/work2/graphcodebert-code-clone-detection/models/graphcodebert-clone-detection",
"dataset_name": "PoolC/1-fold-clone-detection-600k-5fold",
"dataset_split_strategy": {
"train_split": "train",
"heldout_split": "val",
"note": "The repository provides one of 5 predefined folds as `train` + `val`; those groups are disjoint and are kept as-is. `val` is partitioned further into validation/test along problem-group boundaries.",
"heldout_groups": 59,
"validation_groups": 29,
"test_groups": 30,
"dropped_cross_boundary_pairs": 337398,
"validation": {
"num_examples": 483738,
"negatives_label_0": 157558,
"positives_label_1": 326180,
"positive_ratio": 0.674291,
"num_groups": 29
},
"test": {
"num_examples": 503224,
"negatives_label_0": 167224,
"positives_label_1": 336000,
"positive_ratio": 0.667695,
"num_groups": 30
}
},
"num_train_examples": 50000,
"num_validation_examples": 20000,
"num_test_examples": 20000,
"class_distributions": {
"train": {
"num_examples": 50000,
"negatives_label_0": 25000,
"positives_label_1": 25000,
"positive_ratio": 0.5,
"num_groups": 240
},
"validation": {
"num_examples": 20000,
"negatives_label_0": 10000,
"positives_label_1": 10000,
"positive_ratio": 0.5,
"num_groups": 29
},
"test": {
"num_examples": 20000,
"negatives_label_0": 10000,
"positives_label_1": 10000,
"positive_ratio": 0.5,
"num_groups": 30
}
},
"class_weighting": {
"mode": "auto",
"threshold": 0.6,
"majority_class_share": 0.5,
"applied": false,
"weights": null,
"reason": "Measured majority-class share 0.5000 is within the 0.6 threshold, so weighted cross entropy is NOT used."
},
"sequence_length": 512,
"data_flow_length": 128,
"total_sequence_length": 640,
"per_device_train_batch_size": 16,
"gradient_accumulation_steps": 1,
"effective_batch_size": 16,
"learning_rate": 2e-05,
"num_train_epochs": 3.0,
"optimizer": "adamw_torch",
"scheduler": "linear",
"warmup_ratio": 0.1,
"warmup_steps": 938,
"weight_decay": 0.01,
"max_grad_norm": 1.0,
"mixed_precision": "fp16",
"gradient_checkpointing": false,
"seed": 42,
"training_time_seconds": 6921.9,
"training_time_hours": 1.923,
"train_runtime_metrics": {
"train_runtime": 6918.8074,
"train_samples_per_second": 21.68,
"train_steps_per_second": 1.355,
"total_flos": 0.0,
"train_loss": 0.43145505716959637,
"epoch": 3.0
},
"gpu": {
"cuda_available": true,
"torch_version": "2.11.0+cu128",
"transformers_version": "5.17.0",
"python_version": "3.12.3",
"platform": "Linux-6.18.33.2-microsoft-standard-WSL2-x86_64-with-glibc2.39",
"gpu_name": "NVIDIA GeForce RTX 5060 Ti",
"gpu_count": 1,
"gpu_total_memory_gb": 15.9,
"gpu_capability": "12.0",
"cuda_version": "12.8"
},
"parameters": {
"trainable_parameters": 125236994,
"total_parameters": 125236994
},
"best_validation_f1": 0.8671882190520018,
"best_checkpoint": "./outputs/checkpoint-9000",
"validation_metrics": {
"loss": 0.34459105134010315,
"accuracy": 0.8557,
"precision": 0.8032395566922421,
"recall": 0.9422,
"f1": 0.8671882190520018,
"macro_f1": 0.8546121719233737,
"tp": 9422,
"tn": 7692,
"fp": 2308,
"fn": 578,
"confusion_matrix": [
[
7692,
2308
],
[
578,
9422
]
],
"confusion_matrix_layout": "[[TN, FP], [FN, TP]]",
"support": {
"num_examples": 20000,
"label_0": 10000,
"label_1": 10000
},
"runtime": 173.9737,
"samples_per_second": 114.96,
"steps_per_second": 3.592,
"num_examples": 20000,
"eval_seconds": 174.0
},
"test_metrics": {
"loss": 0.3165612816810608,
"model_preparation_time": 0.0022,
"accuracy": 0.87465,
"precision": 0.8409939018840448,
"recall": 0.924,
"f1": 0.8805450993472149,
"macro_f1": 0.874343974488208,
"tp": 9240,
"tn": 8253,
"fp": 1747,
"fn": 760,
"confusion_matrix": [
[
8253,
1747
],
[
760,
9240
]
],
"confusion_matrix_layout": "[[TN, FP], [FN, TP]]",
"support": {
"num_examples": 20000,
"label_0": 10000,
"label_1": 10000
},
"runtime": 181.6455,
"samples_per_second": 110.105,
"steps_per_second": 3.441,
"num_examples": 20000,
"eval_seconds": 181.7
},
"preprocessing": {
"snippet_pool": {
"num_unique_snippets": 44950,
"scan_seconds": 46.3,
"snippets_with_multiple_groups": 1,
"cross_split_snippet_overlap": {
"train|val:snippets": 0,
"train|val:groups": 0
},
"per_split": {
"train": {
"num_examples": 5388622,
"negatives_label_0": 2694311,
"positives_label_1": 2694311,
"positive_ratio": 0.5,
"num_groups": 240
},
"val": {
"num_examples": 1324360,
"negatives_label_0": 662180,
"positives_label_1": 662180,
"positive_ratio": 0.5,
"num_groups": 59
}
}
},
"feature_extraction": {
"num_snippets": 44950,
"extraction_seconds": 64.1,
"status_counts": {
"ok": 44930,
"comment_strip_failed": 13,
"dfg_failed": 7
},
"snippets_with_empty_dataflow": 263,
"dataflow_nodes_mean": 44.22,
"dataflow_nodes_p50": 33,
"dataflow_nodes_p95": 127,
"dataflow_nodes_max": 193,
"code_tokens_mean": 137.9,
"code_tokens_truncated": 885,
"total_dataflow_edges": 2481388,
"sequence_length": 640
},
"subsampling": {
"train": {
"subsampled": true,
"kept": 50000,
"balanced": true
},
"validation": {
"subsampled": true,
"kept": 20000,
"balanced": true
},
"test": {
"subsampled": true,
"kept": 20000,
"balanced": true
}
}
},
"sanity_check": {
"passed": true,
"device": "cuda",
"checks": [
{
"check": "1_dataset_loading",
"passed": true,
"detail": "train=50000 val=20000 test=20000"
},
{
"check": "2_column_detection",
"passed": true,
"detail": "code columns=code1/code2, label=similar; excluded from features: ['code1_group', 'code2_group', 'pair_id', 'question_pair_id']"
},
{
"check": "3_label_correctness",
"passed": true,
"detail": "labels in {0,1}; validation positive ratio=0.5"
},
{
"check": "4_tokenisation",
"passed": true,
"detail": "<s>...</s> wrapping OK, 93 code tokens, decodes to 'import math a , b , c = map ( int , input ( ) . split ( ) ) '"
},
{
"check": "5_dataflow_extraction",
"passed": true,
"detail": "44687/44950 snippets have a non-empty data-flow graph, 2481388 edges, mean nodes=44.22"
},
{
"check": "6_attention_mask",
"passed": true,
"detail": "shape=(640, 640) (code_length 512 + data_flow_length 128), 93 code tokens, 30 nodes, density=0.0215, padding rows empty"
},
{
"check": "7_forward_pass",
"passed": true,
"detail": "logits shape=(16, 2) for batch of 16 pairs; mask tensor=(16, 640, 640)"
},
{
"check": "8_finite_loss",
"passed": true,
"detail": "loss=0.6978 (chance level ~0.6931)"
},
{
"check": "9_training_step",
"passed": true,
"detail": "loss=0.6798, grad_norm=2.4929, weights updated (amp=on, dtype=torch.float16)"
}
],
"parameters": {
"trainable_parameters": 125236994,
"total_parameters": 125236994
},
"sequence_length": 640
},
"test_metrics_provenance": {
"note": "Re-scored after fixing a transformers v5 dataloader-caching bug: Trainer.get_eval_dataloader caches under the key 'eval' when a Dataset object is passed and dataloader_persistent_workers=True, so the original in-training test evaluation silently re-scored the validation split. evaluate_split now uses Trainer.predict and verifies the returned labels.",
"superseded_test_f1": 0.8671882190520018,
"corrected_test_f1": 0.8805450993472149,
"validation_f1_unchanged": 0.8671882190520018
}
}