#![allow(clippy::doc_markdown)] //! EveryCycle inference runtime. //! //! Loads quantized models, places layers across heterogeneous compute (GPU, //! CPU, NVMe overflow), drives the forward pass, and streams tokens to the //! scheduler. //! //! See `docs/architecture.md` in the workspace root for the broader sketch. pub mod executor; pub use executor::{ ActivationHandle, BoundDevice, Dtype, Executor, ExecutorError, PortableActivation, SubGraphPlan, TensorShape, Vendor, };