tensorfold serve "$MODEL" \ --served-name qwen233-a35b \ --resident-budget 39GiB \ --loader-backend native \ --host 127.0.0.1 \ --port 8421 GPU stays 1- -max- 8 % (most of the time closer to 1 %)
tensorfold serve "$MODEL"
--served-name qwen233-a35b
--resident-budget 39GiB
--loader-backend native
--host 127.0.0.1
--port 8421
GPU stays 1- -max- 8 % (most of the time closer to 1 %)