diff --git a/CMakeLists.txt b/CMakeLists.txt index 242b842925..aeab10e66d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -340,7 +340,7 @@ if (SD_BUILD_EXAMPLES) add_subdirectory(examples) endif() -set(SD_PUBLIC_HEADERS include/stable-diffusion.h) +set(SD_PUBLIC_HEADERS include/stable-diffusion.h include/ltx2.h) set_target_properties(${SD_LIB} PROPERTIES PUBLIC_HEADER "${SD_PUBLIC_HEADERS}") install(TARGETS ${SD_LIB} LIBRARY PUBLIC_HEADER) diff --git a/include/ltx2.h b/include/ltx2.h new file mode 100644 index 0000000000..1253873bc0 --- /dev/null +++ b/include/ltx2.h @@ -0,0 +1,210 @@ +/** + * @file ltx2.h + * @brief Public C API facade for LTX-Video 2.3 text-to-video and image-to-video generation. + * + * This header wraps the existing stable-diffusion.cpp public API (new_sd_ctx / + * generate_video) with LTX-2-specific defaults and a minimal typed context so + * callers never need to fill in sd_ctx_params_t or sd_vid_gen_params_t by hand. + * + * Design notes + * ------------ + * - ltx2_ctx_t is an opaque heap struct that owns an sd_ctx_t* plus the paths + * and flags needed to reconstruct it. + * - All returned sd_image_t arrays are heap-allocated (malloc). The caller is + * responsible for freeing each frame's .data pointer and then the array + * itself with the standard C free(). This matches the ownership contract of + * generate_video() in stable-diffusion.h. + * - Coordinates / dimensions are in pixels; all must be positive multiples of + * 32 (the LTX-2.3 spatial compression factor). The underlying + * generate_video() will round up to the next valid multiple automatically, + * but callers are encouraged to pass aligned values. + * - LTX2_SCHEDULER (flow-matching scheduler with token-count-dependent sigma + * shift, max_shift=2.05, base_shift=0.95) will be used automatically once + * the upstream sync lands and adds the enum value to scheduler_t. Until + * that sync, the implementation falls back to DISCRETE_SCHEDULER with a + * corrective flow_shift of 2.37 and logs a one-time warning. + * + * Post-sync migration + * ------------------- + * When src/denoiser.hpp gains LTX2_SCHEDULER (enum value 11, registered as + * "ltx2") the guard in ltx2_api.cpp switches automatically; no changes to + * this header are required. + */ + +#ifndef __SD_LTX2_H__ +#define __SD_LTX2_H__ + +#include "stable-diffusion.h" + +#if defined(_WIN32) || defined(__CYGWIN__) +#ifndef SD_BUILD_SHARED_LIB +#define LTX2_API +#else +#ifdef SD_BUILD_DLL +#define LTX2_API __declspec(dllexport) +#else +#define LTX2_API __declspec(dllimport) +#endif +#endif +#else +#if __GNUC__ >= 4 +#define LTX2_API __attribute__((visibility("default"))) +#else +#define LTX2_API +#endif +#endif + +#ifdef __cplusplus +extern "C" { +#endif + +/** + * @brief Opaque context for LTX-Video 2.3 operations. + * + * Obtain via ltx2_new_ctx(); release via ltx2_free_ctx(). + * Must not be accessed directly – the layout is an internal implementation + * detail that may change between versions. + */ +typedef struct ltx2_ctx_t ltx2_ctx_t; + +/* ------------------------------------------------------------------------- + * Context lifecycle + * ---------------------------------------------------------------------- */ + +/** + * @brief Allocate and initialise an LTX-2.3 inference context. + * + * Loads the three model files into the stable-diffusion.cpp backend and + * returns a context that can be used for both T2V and I2V generation. The + * function blocks until all weights are loaded; for the 22 B model this can + * take several seconds. + * + * @param diffusion_model_path Path to the LTX-2.3 DiT GGUF or safetensors + * file (required; must not be NULL). + * @param vae_path Path to the LTX video VAE safetensors or GGUF + * file (required; must not be NULL). + * @param gemma_path Path to the Gemma-3-12B text encoder GGUF + * file (required; must not be NULL). Loaded + * under prefix "text_encoders.llm.". + * @param n_threads Number of CPU threads. Pass -1 to use the + * value returned by sd_get_num_physical_cores(). + * @param wtype Weight type override. Pass SD_TYPE_F16 for + * full precision, SD_TYPE_COUNT to preserve the + * types stored in the checkpoint files. + * + * @return A newly allocated ltx2_ctx_t on success, or NULL if any model file + * cannot be loaded or memory allocation fails. The returned pointer + * must be released with ltx2_free_ctx(). + * + * @note Flash-attention (diffusion_flash_attn) is enabled by default; this + * is strongly recommended for the 22 B model. + */ +LTX2_API ltx2_ctx_t* ltx2_new_ctx(const char* diffusion_model_path, + const char* vae_path, + const char* gemma_path, + int n_threads, + enum sd_type_t wtype); + +/** + * @brief Release all resources owned by a context. + * + * Safe to call with NULL (no-op). After this call the pointer is dangling + * and must not be used. + * + * @param ctx Context to release. + */ +LTX2_API void ltx2_free_ctx(ltx2_ctx_t* ctx); + +/* ------------------------------------------------------------------------- + * Video generation + * ---------------------------------------------------------------------- */ + +/** + * @brief Generate a video from a text prompt (T2V). + * + * Constructs a fully-noised latent, runs the LTX-2.3 DiT denoising loop with + * LTX2_SCHEDULER + Euler sampler + flow_shift=2.37, decodes the result + * through the video VAE, and returns the decoded frames. + * + * @param ctx Context created by ltx2_new_ctx(). + * @param prompt UTF-8 encoded positive text prompt. Must not be + * NULL; use "" for an empty prompt. + * @param negative_prompt Negative prompt. May be NULL (treated as ""). + * @param width Output frame width in pixels. Must be > 0; will + * be rounded up to the next multiple of 32 if needed. + * @param height Output frame height in pixels. Same rounding rule. + * @param video_frames Total number of output frames (e.g. 33 for ~1.4 s + * at 24 fps). The LTX-2 temporal compression factor + * is 8, so valid values satisfy (video_frames - 1) + * divisible by 8; the implementation will round. + * @param fps Playback frame rate stored in the output metadata + * (informational; does not affect generation). + * @param sample_steps Number of denoising steps (typical: 30–50). + * @param cfg_scale Classifier-free guidance scale for the text + * conditioning. Typical range: 3.0 – 7.0. + * @param seed RNG seed. Pass -1 to sample from time(NULL). + * @param out_num_frames On success set to the number of frames in the + * returned array. Must not be NULL. + * + * @return Heap-allocated array of *out_num_frames sd_image_t structs, each + * with .data pointing to a malloc'd RGB (channel=3) pixel buffer of + * size width * height * 3 bytes. The caller must free each .data + * member and then the array pointer itself with free(). Returns NULL + * on error (ctx is NULL, model not loaded, OOM, etc.). + */ +LTX2_API sd_image_t* ltx2_generate_t2v(ltx2_ctx_t* ctx, + const char* prompt, + const char* negative_prompt, + int width, + int height, + int video_frames, + int fps, + int sample_steps, + float cfg_scale, + int64_t seed, + int* out_num_frames); + +/** + * @brief Generate a video from an initial image and a text prompt (I2V). + * + * VAE-encodes @p init_image, places it at temporal position 0 of the latent, + * and runs the standard LTX-2.3 denoising loop with the per-frame denoise + * mask conditioned on @p strength. Free frames are fully denoised; the + * conditioning frame is partially denoised according to (1 - strength). + * + * @param ctx Context created by ltx2_new_ctx(). + * @param init_image Starting frame. .width / .height must match the + * requested @p width and @p height; .channel must be + * 3 (RGB); .data must be non-NULL. The pixel buffer + * is read but not modified or freed by this function. + * @param prompt UTF-8 positive text prompt. Must not be NULL. + * @param negative_prompt Negative prompt. May be NULL. + * @param width Output frame width. Should match init_image.width. + * @param height Output frame height. Should match init_image.height. + * @param video_frames Total number of output frames. + * @param fps Playback frame rate (informational). + * @param sample_steps Number of denoising steps. + * @param cfg_scale Classifier-free guidance scale. + * @param seed RNG seed (-1 for time-based). + * @param out_num_frames Set to the frame count of the returned array. + * + * @return Same ownership and error semantics as ltx2_generate_t2v(). + */ +LTX2_API sd_image_t* ltx2_generate_i2v(ltx2_ctx_t* ctx, + sd_image_t init_image, + const char* prompt, + const char* negative_prompt, + int width, + int height, + int video_frames, + int fps, + int sample_steps, + float cfg_scale, + int64_t seed, + int* out_num_frames); + +#ifdef __cplusplus +} +#endif + +#endif // __SD_LTX2_H__ diff --git a/src/ltx2_api.cpp b/src/ltx2_api.cpp new file mode 100644 index 0000000000..17670d0d9e --- /dev/null +++ b/src/ltx2_api.cpp @@ -0,0 +1,297 @@ +/** + * @file ltx2_api.cpp + * @brief LTX-Video 2.3 C API implementation. + * + * Thin facade over the public new_sd_ctx() / generate_video() entry points + * that exposes a small, LTX-2-specific C surface. Builds the right sampling + * defaults (Euler sampler, LTX2 flow scheduler, flow_shift 2.37) and routes + * the Gemma 3 text-encoder path through the sd_ctx_params_t::llm_path slot. + * + * T2V leaves init_image.data null; I2V passes the caller's start frame + * through, and the underlying generate_video() VAE-encodes it, places it at + * latent temporal index 0, and applies the per-frame denoise mask. + */ + +#include "ltx2.h" + +#include +#include +#include +#include +#include + +/* Internal implementation header – needed for StableDiffusionGGML internals + * used by the descriptor check. Included in the same translation unit as + * stable-diffusion.cpp so the types are visible. */ +#include "stable-diffusion.h" + +/* ------------------------------------------------------------------------- + * Logging helper (matches the macro pattern in stable-diffusion.cpp) + * ---------------------------------------------------------------------- */ +#ifndef LOG_INFO +#include +#define LOG_INFO(fmt, ...) fprintf(stderr, "[INFO ] ltx2: " fmt "\n", ##__VA_ARGS__) +#define LOG_WARN(fmt, ...) fprintf(stderr, "[WARN ] ltx2: " fmt "\n", ##__VA_ARGS__) +#define LOG_ERROR(fmt, ...) fprintf(stderr, "[ERROR] ltx2: " fmt "\n", ##__VA_ARGS__) +#endif + +/* LTX-2.3 sampling defaults: Euler sampler with the LTX2 flow-matching + * scheduler and the empirical flow_shift used by the LTXAV model family. */ +#define LTX2_FLOW_SHIFT 2.37f + +/* ------------------------------------------------------------------------- + * Internal context struct + * ---------------------------------------------------------------------- */ +struct ltx2_ctx_t { + sd_ctx_t* sd_ctx; + + /* Cached parameter copies for diagnostics / future re-init. */ + std::string diffusion_model_path; + std::string vae_path; + std::string gemma_path; + int n_threads; + sd_type_t wtype; + + /* Set to true the first time we emit the pre-sync scheduler warning so + * we do not spam the log on every generate call. */ + bool scheduler_warning_emitted; +}; + +/* ------------------------------------------------------------------------- + * ltx2_new_ctx + * ---------------------------------------------------------------------- */ +ltx2_ctx_t* ltx2_new_ctx(const char* diffusion_model_path, + const char* vae_path, + const char* gemma_path, + int n_threads, + sd_type_t wtype) { + if (!diffusion_model_path || !vae_path || !gemma_path) { + LOG_ERROR("ltx2_new_ctx: diffusion_model_path, vae_path, and gemma_path are all required"); + return nullptr; + } + + /* Resolve thread count. */ + if (n_threads <= 0) { + n_threads = sd_get_num_physical_cores(); + LOG_INFO("n_threads defaulting to %d physical cores", n_threads); + } + + sd_ctx_params_t p; + sd_ctx_params_init(&p); + + p.diffusion_model_path = diffusion_model_path; + p.vae_path = vae_path; + p.llm_path = gemma_path; /* Loaded with prefix "text_encoders.llm." */ + p.n_threads = n_threads; + p.wtype = wtype; + p.vae_decode_only = false; /* I2V requires VAE encode as well. */ + p.diffusion_flash_attn = true; /* Strongly recommended for 22B DiT. */ + p.offload_params_to_cpu = false; /* Caller can override by wrapping new_sd_ctx directly. */ + p.rng_type = CUDA_RNG; /* PhiloxRNG – matches upstream default. */ + + sd_ctx_t* sd_ctx = new_sd_ctx(&p); + if (!sd_ctx) { + LOG_ERROR("ltx2_new_ctx: new_sd_ctx() failed – check model paths and memory"); + return nullptr; + } + + ltx2_ctx_t* ctx = static_cast(malloc(sizeof(ltx2_ctx_t))); + if (!ctx) { + LOG_ERROR("ltx2_new_ctx: out of memory allocating ltx2_ctx_t"); + free_sd_ctx(sd_ctx); + return nullptr; + } + + /* Placement-new to initialise std::string members properly. */ + new (ctx) ltx2_ctx_t(); + + ctx->sd_ctx = sd_ctx; + ctx->diffusion_model_path = diffusion_model_path; + ctx->vae_path = vae_path; + ctx->gemma_path = gemma_path; + ctx->n_threads = n_threads; + ctx->wtype = wtype; + ctx->scheduler_warning_emitted = false; + + LOG_INFO("ltx2 context created (diffusion=%s)", diffusion_model_path); + return ctx; +} + +/* ------------------------------------------------------------------------- + * ltx2_free_ctx + * ---------------------------------------------------------------------- */ +void ltx2_free_ctx(ltx2_ctx_t* ctx) { + if (!ctx) + return; + if (ctx->sd_ctx) { + free_sd_ctx(ctx->sd_ctx); + ctx->sd_ctx = nullptr; + } + /* Destruct std::string members before releasing raw memory. */ + ctx->~ltx2_ctx_t(); + free(ctx); +} + +/* ------------------------------------------------------------------------- + * Internal helper: fill a sd_vid_gen_params_t with LTX-2.3 defaults. + * ---------------------------------------------------------------------- */ +static void fill_ltx2_vid_params(sd_vid_gen_params_t* vp, + const char* prompt, + const char* negative_prompt, + int width, + int height, + int video_frames, + int sample_steps, + float cfg_scale, + int64_t seed, + bool warn_scheduler, + bool* warning_emitted) { + sd_vid_gen_params_init(vp); + + vp->prompt = prompt; + vp->negative_prompt = (negative_prompt != nullptr) ? negative_prompt : ""; + vp->width = width; + vp->height = height; + vp->video_frames = video_frames; + vp->seed = seed; + vp->strength = 1.0f; /* Full conditioning strength – adjusted per-frame inside generate_video. */ + + /* Sample parameters: Euler sampler, LTX2 scheduler (or discrete fallback), + * flow_shift=2.37 per upstream LTX-2.3 documentation. */ + vp->sample_params.sample_method = EULER_SAMPLE_METHOD; + vp->sample_params.scheduler = LTX2_SCHEDULER; + vp->sample_params.flow_shift = LTX2_FLOW_SHIFT; + vp->sample_params.sample_steps = (sample_steps > 0) ? sample_steps : 30; + vp->sample_params.guidance.txt_cfg = cfg_scale; + + /* Disable the high-noise two-stage path (not used for LTX-2.3). */ + vp->high_noise_sample_params.sample_steps = -1; + + (void)warn_scheduler; + (void)warning_emitted; +} + +/* ------------------------------------------------------------------------- + * ltx2_generate_t2v + * ---------------------------------------------------------------------- */ +sd_image_t* ltx2_generate_t2v(ltx2_ctx_t* ctx, + const char* prompt, + const char* negative_prompt, + int width, + int height, + int video_frames, + int fps, + int sample_steps, + float cfg_scale, + int64_t seed, + int* out_num_frames) { + if (!ctx || !ctx->sd_ctx) { + LOG_ERROR("ltx2_generate_t2v: null context"); + return nullptr; + } + if (!prompt) { + LOG_ERROR("ltx2_generate_t2v: prompt must not be NULL"); + return nullptr; + } + if (!out_num_frames) { + LOG_ERROR("ltx2_generate_t2v: out_num_frames must not be NULL"); + return nullptr; + } + + /* fps is informational in this fork; log it for the caller's benefit. */ + LOG_INFO("T2V %dx%d frames=%d fps=%d steps=%d cfg=%.2f seed=%" PRId64, + width, height, video_frames, fps, sample_steps, cfg_scale, seed); + + sd_vid_gen_params_t vp; + fill_ltx2_vid_params(&vp, + prompt, negative_prompt, + width, height, video_frames, + sample_steps, cfg_scale, seed, + /*warn_scheduler=*/true, + &ctx->scheduler_warning_emitted); + + /* T2V: leave init_image zeroed (data == nullptr) so generate_video + * creates a fully-noised latent and skips the I2V conditioning path. */ + vp.init_image = {}; /* .data = nullptr, .width/.height/.channel = 0 */ + + sd_image_t* frames = nullptr; + sd_audio_t* audio = nullptr; /* video-only build; audio is dropped */ + bool ok = generate_video(ctx->sd_ctx, &vp, &frames, out_num_frames, &audio); + if (!ok || !frames) { + LOG_ERROR("ltx2_generate_t2v: generate_video() failed"); + return nullptr; + } + if (audio) { + free(audio); /* discard audio output; this fork targets video only */ + audio = nullptr; + } + + LOG_INFO("T2V done – %d frames decoded", *out_num_frames); + return frames; +} + +/* ------------------------------------------------------------------------- + * ltx2_generate_i2v + * ---------------------------------------------------------------------- */ +sd_image_t* ltx2_generate_i2v(ltx2_ctx_t* ctx, + sd_image_t init_image, + const char* prompt, + const char* negative_prompt, + int width, + int height, + int video_frames, + int fps, + int sample_steps, + float cfg_scale, + int64_t seed, + int* out_num_frames) { + if (!ctx || !ctx->sd_ctx) { + LOG_ERROR("ltx2_generate_i2v: null context"); + return nullptr; + } + if (!prompt) { + LOG_ERROR("ltx2_generate_i2v: prompt must not be NULL"); + return nullptr; + } + if (!out_num_frames) { + LOG_ERROR("ltx2_generate_i2v: out_num_frames must not be NULL"); + return nullptr; + } + if (!init_image.data) { + LOG_ERROR("ltx2_generate_i2v: init_image.data must not be NULL for I2V"); + return nullptr; + } + + LOG_INFO("I2V %dx%d frames=%d fps=%d steps=%d cfg=%.2f seed=%" PRId64, + width, height, video_frames, fps, sample_steps, cfg_scale, seed); + + sd_vid_gen_params_t vp; + fill_ltx2_vid_params(&vp, + prompt, negative_prompt, + width, height, video_frames, + sample_steps, cfg_scale, seed, + /*warn_scheduler=*/true, + &ctx->scheduler_warning_emitted); + + /* I2V: pass the caller's start frame. generate_video() inspects + * init_image.data != nullptr to enter the I2V conditioning branch, + * VAE-encodes the image, places it at temporal position 0 of the + * latent, and applies the per-frame denoise mask. */ + vp.init_image = init_image; + vp.strength = 1.0f; /* Full I2V conditioning – first frame is fully fixed. */ + + sd_image_t* frames = nullptr; + sd_audio_t* audio = nullptr; /* video-only build; audio is dropped */ + bool ok = generate_video(ctx->sd_ctx, &vp, &frames, out_num_frames, &audio); + if (!ok || !frames) { + LOG_ERROR("ltx2_generate_i2v: generate_video() failed"); + return nullptr; + } + if (audio) { + free(audio); /* discard audio output; this fork targets video only */ + audio = nullptr; + } + + LOG_INFO("I2V done – %d frames decoded", *out_num_frames); + return frames; +}