recICL / config.json
devtaji's picture
recICL v0
08646a9
Raw History Blame Contribute Delete
7.27 kB
{
"name": "recICL",
"model_type": "icl-recommender",
"architecture": "RecPFN",
"architecture_source": "https://github.com/SAP-samples/tabular-ai-recpfn (upstream commit 1af320c, Apache-2.0)",
"paper": "arXiv:2608.19735 (SIGIR '26), doi:10.1145/3805712.3809696",
"embedding_dim": 1536,
"n_parameters": 169918464,
"dtype": "float32",
"model_config": {
"n_layers": 4,
"n_heads": 8,
"dropout": 0.2,
"icl_module_type": "alternating",
"positional_embedding_scheme": "hard-alibi",
"max_sequence_len": 15,
"train_with_icl": true,
"num_icl_examples": 8,
"icl_k": 2,
"weighted_icl_sampling": true,
"weighted_decay_rate": 1.0,
"normalize_embs": true,
"llm": "qwen1b",
"baseline_config": "balanced"
},
"inference": {
"history_items_used": 14,
"context_sequence_max_len": 15,
"num_context_sequences": 8,
"context_retrieval": "SAP InvertedIndexMatcher: pool sequences cut into windows of 15 (step 7); candidates share one of the query's last 2 items; top 8 by recency-weighted Jaccard (decay 1.0)",
"score": "dot product of the output at the last history position with each L2-normalised item embedding"
},
"item_embeddings": {
"model_id": "Alibaba-NLP/gte-Qwen2-1.5B-instruct",
"revision": "a9af15a6372d7d6b25e9fb07c2ccb9e1fe645644",
"dim": 1536,
"trust_remote_code": true,
"compute_dtype": "bf16",
"max_seq_length": 512,
"prompt": null,
"pooling": "last token (<|endoftext|> appended by the remote-code tokenizer), right padding",
"normalization": "L2 in fp32",
"amazon_item_text_format": "Title: {title}; Brand: {brand}; Categories: {c1, c2, ...}"
},
"training": {
"code": "SAP-samples/tabular-ai-recpfn src/ at 1af320c + infrastructure patches (resume, logging, NaN diagnostics; opt-in fixes all OFF) - training math as released",
"config": {
"train_with_icl": true,
"num_icl_examples": 128,
"train_with_synthetic_data": true,
"n_layers": 4,
"batch_size": 16,
"reg_lambda": 0.0001,
"warmup_epochs": 6,
"baseline_config": "balanced",
"learning_rate": 0.0001,
"early_stopping_patience": 20,
"positional_embedding_scheme": "hard-alibi",
"optimizer": "adamw",
"llm": "qwen1b",
"max_sequence_len": 15,
"num_epochs": 120,
"use_cached_embeddings": true,
"num_gradient_accumulation_steps": 2,
"early_stopping_metric": "hr@10",
"loss": "sce",
"train_epoch_size": 500,
"val_epoch_size": 200,
"num_sdg_items": 1000,
"normalize_embs": true,
"n_heads": 8,
"dropout": 0.2,
"icl_module_type": "alternating",
"val_k": 10
},
"training_seed": 0,
"optimizer": "AdamW (torch defaults besides lr), linear warmup for warmup_epochs, then cosine annealing to 0.01*lr",
"data": "synthetic sequences only (no real user data); each batch = one synthetic environment of 1000 items whose embeddings are sampled from the embedding store below",
"embedding_store": {
"rows": 800000,
"composition": {
"fiqa": 64248,
"quora": 200000,
"fever": 200000,
"msmarco": 335752
},
"sources": "BEIR (FiQA-2018, Quora, FEVER, MS MARCO), sampled with seed 0",
"encoder": "same as item_embeddings",
"sha256": "fb9374462837c1b41188d15d50c55faf04096373f918ea31be57ddcdd6a29620"
},
"stages": {
"stage1": {
"synthetic_prior": "random graph",
"sdg_params": {
"transition_matrix_prior": [
"random_graph"
],
"popularity_bias": [
0.0,
0.1,
0.2,
0.3,
0.5
],
"window_size": [
1,
2,
3
],
"decay_coeff": [
0.5,
0.8
],
"repeated_items": [
true
]
},
"init": "random (seed 0)",
"epochs_run": 120,
"best_epoch": 101,
"early_stopped": false,
"best_synthetic_ratio_hr@10": 7.974883964010977,
"gpu": [
"NVIDIA H200"
],
"wall_hours": 2.573,
"first_epoch_end": "2026-09-29T07:33:10Z",
"finished": "2026-09-29T10:05:54Z"
},
"stage2": {
"synthetic_prior": "random graph + latent factor",
"sdg_params": {
"transition_matrix_prior": [
"random_graph",
"latent_factor"
],
"num_latent_concepts": [
20,
40,
60
],
"mean_num_concepts_per_item": [
2,
3,
5
],
"mean_num_concepts_per_user": [
2,
3,
5
],
"popularity_bias": [
0.0,
0.1,
0.2
],
"window_size": [
1,
2,
3
],
"decay_coeff": [
0.5,
0.8
],
"logit_scale": [
9.0,
12.0
],
"concept_matrix_diagonal_deviation_rate": [
0.0,
0.1,
0.3
],
"repeated_items": [
true
],
"mean_num_related_items": [
1,
3,
5,
7
],
"repeated_item_probability": [
0.1,
0.2,
0.3
]
},
"init": "stage-1 best checkpoint (epoch 101)",
"epochs_run": 47,
"best_epoch": 27,
"early_stopped": true,
"best_synthetic_ratio_hr@10": 3.9308144874246915,
"gpu": [
"NVIDIA H100 80GB HBM3"
],
"wall_hours": 1.099,
"first_epoch_end": "2026-09-29T10:07:51Z",
"finished": "2026-09-29T11:11:55Z"
}
},
"early_stopping": "hr@10 ratio (model / EmbKNN-balanced) on synthetic validation batches, patience 20",
"selection": {
"note": "best of our 9 training runs by mean validation HR@10 over the 7 Amazon validation splits",
"amazon7_val_hr@10": 0.13147873010138936,
"amazon7_val_hr@10_rank": 1,
"amazon7_val_mrr@10": 0.0884027748063644,
"amazon7_val_mrr@10_rank": 2,
"n_runs": 9
}
},
"evaluation_protocol": {
"split": "user-level 70/10/20, seed 42; at most 10,000 users sampled into the test split, then users with a single interaction dropped",
"target": "each test user's last item; input = up to 14 preceding items",
"context": "8 sequences from train-split users (retrieval as in 'inference')",
"batch_size": 1,
"ranking": "full item catalog, no candidate sampling, repeated items allowed",
"metrics": [
"HR@10",
"MRR@10"
],
"hardware": "NVIDIA L4, fp32, TF32 off"
},
"checkpoint": {
"source": "release_k4_s0/stage2/model.pth",
"source_sha256": "661fd3415a0b8f2d727898006e6314dcd273e15f59da3383668f7bc1a3997a3a",
"safetensors_sha256": "02c4edc3a4944a9e7a0b6025217b3ea7af16e53c9483e85e9a573074a228a60c",
"safetensors_bytes": 679680696,
"n_tensors": 64,
"src_fingerprint": "6c3ca6431e028e27"
}
}