Quickstart & Execution Recipes

Standard usage patterns and rapid prototyping code

Basic Execution Recipe

from termux_bitnet import BitNetEngine, BitNetConfig

# 1. Initialize engine with model wrapping and GPU acceleration options
config = BitNetConfig(
    model_path="~/.cache/termux-bitnet/models/falcon-e-1b-instruct-i2_s.gguf",
    device="gpu",           # "auto", "cpu", or "gpu" (Vulkan via ameva-runtime)
    n_gpu_layers=24,        # Offload 24 layers to GPU
    chat_template="falcon", # Auto-wrapping: ChatML, Falcon, or Raw
    chunk_layers=4,         # Prevent Mali GPU watchdog timeout
    vocab_slice=32768,      # Save 576MB VRAM on LM Head
    n_threads=4,
    temperature=0.7,
    top_p=0.95
)

# 2. Stream tokens in real time (up to 34.35 tok/s on S25, 4.46 tok/s on A53)
with BitNetEngine(config) as engine:
    print("[Prompt]: What is the capital of France?")
    print("[Response]: ", end="", flush=True)
    for token in engine.generate_stream("What is the capital of France?"):
        print(token, end="", flush=True)
    print()
    metrics = engine.get_last_metrics()
    print(f"Speed: {metrics.tokens_per_second:.2f} tok/s | Prompt tokens: {metrics.prompt_tokens}")