mirror of
https://github.com/NicolasBohn/NexQuant.git
synced 2026-08-08 12:37:44 +00:00
feat: run Kronos on CPU to avoid GPU conflict with llama-server
This commit is contained in:
@@ -329,13 +329,9 @@ def quant(
|
||||
threading.Thread(target=start_cli_dash, daemon=True).start()
|
||||
time.sleep(1)
|
||||
|
||||
# ---- Kronos Factor: skip if GPU unavailable (CUDA OOM with llama-server) ----
|
||||
# ---- Kronos Factor: CPU inference to avoid GPU conflict with llama-server ----
|
||||
try:
|
||||
import torch
|
||||
if torch.cuda.is_available() and torch.cuda.get_device_properties(0).total_memory > 20 * 1024**3:
|
||||
_ensure_kronos_factor_in_pool(console)
|
||||
else:
|
||||
console.print("[dim]Kronos Factor skipped — GPU < 20GB or CUDA unavailable[/dim]")
|
||||
_ensure_kronos_factor_in_pool(console)
|
||||
except Exception:
|
||||
console.print("[dim]Kronos Factor skipped — torch not available[/dim]")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user