| from nzfc_gram_runtime import NZFCGramLongMemoryChat |
| from nzfc_gram_runtime.nonquant import attach_nonquant_gemma |
| from nzfc_gram_runtime.cache_profiles import attach_adaptive_kv_cache_generation |
| from nzfc_gram_runtime.quality import attach_answer_quality_governor |
|
|
| MODEL_ID = 'google/gemma-4-E2B-it' |
|
|
| bot = NZFCGramLongMemoryChat( |
| repo_dir='.', |
| model_id=MODEL_ID, |
| memory_db_path='./user_memory_adaptive_cache.sqlite3', |
| load_model=False, |
| require_model=False, |
| preload_static_memory=True, |
| ) |
|
|
| attach_nonquant_gemma( |
| bot, |
| model_id=MODEL_ID, |
| device_map='balanced_low_0', |
| gpu_max_memory_gib=11, |
| gpu_max_memory_gib_candidates=[11, 10, 9, 8], |
| cpu_max_memory_gib=48, |
| prefer_bf16=True, |
| use_fp32=False, |
| ) |
|
|
| attach_adaptive_kv_cache_generation(bot, default_cache_policy='adaptive', default_prefer_cache=True) |
| attach_answer_quality_governor(bot) |
|
|
| out = bot.generate_answer( |
| system_prompt='You are a concise assistant.', |
| user_prompt='Explain memory as evidence in one sentence.', |
| max_new_tokens=80, |
| cache_policy='adaptive', |
| prefer_cache=True, |
| ) |
|
|
| print(out['answer']) |
| print('use_cache:', out.get('use_cache')) |
| print('fallback_used:', out.get('fallback_used')) |
| print('latency_s:', out.get('latency_s')) |
|
|