Quickstart & Execution Recipes
Standard usage patterns and rapid prototyping code
Basic Execution Recipe
from termux_bitnet import BitNetEngine, BitNetConfig
# 1. Initialize engine with model wrapping and GPU acceleration options
config = BitNetConfig(
model_path="~/.cache/termux-bitnet/models/falcon-e-1b-instruct-i2_s.gguf",
device="gpu", # "auto", "cpu", or "gpu" (Vulkan via ameva-runtime)
n_gpu_layers=24, # Offload 24 layers to GPU
chat_template="falcon", # Auto-wrapping: ChatML, Falcon, or Raw
chunk_layers=4, # Prevent Mali GPU watchdog timeout
vocab_slice=32768, # Save 576MB VRAM on LM Head
n_threads=4,
temperature=0.7,
top_p=0.95
)
# 2. Stream tokens in real time (up to 34.35 tok/s on S25, 4.46 tok/s on A53)
with BitNetEngine(config) as engine:
print("[Prompt]: What is the capital of France?")
print("[Response]: ", end="", flush=True)
for token in engine.generate_stream("What is the capital of France?"):
print(token, end="", flush=True)
print()
metrics = engine.get_last_metrics()
print(f"Speed: {metrics.tokens_per_second:.2f} tok/s | Prompt tokens: {metrics.prompt_tokens}")