{"artifactMounts":{"inputField":"_appnz_artifacts_mounts","mode":"virtual-r2","mountPath":"/mnt/appnz/artifacts","supported":true},"comfyBridge":{"node":"AppNZAnyCog","templateRef":"template:{template_name}","workflow":"/api/comfy/spaces/cog-pocket-tts-autoscale/download"},"defaultHardware":"gpu-rtx4090","defaultIdleSecs":120,"hardware":[{"cc":"8.0","gpu":"A100-80GB","id":"gpu-a100","label":"A100 80GB","priceHourText":"$2.62","ramGb":80,"vramGb":80},{"cc":"9.0","gpu":"H100-80GB","id":"gpu-h100","label":"H100 80GB","priceHourText":"$4.78","ramGb":80,"vramGb":80},{"cc":"9.0","gpu":"H200-141GB","id":"gpu-h200","label":"H200 141GB","priceHourText":"$6.38","ramGb":141,"vramGb":141},{"cc":"8.6","gpu":"A40","id":"gpu-a40","label":"A40 48GB","priceHourText":"$0.64","ramGb":48,"vramGb":48},{"cc":"8.9","gpu":"L40S","id":"gpu-l40s","label":"L40S","priceHourText":"$1.38","ramGb":48,"vramGb":48},{"cc":"8.6","gpu":"RTX-3090","id":"gpu-rtx3090","label":"RTX 3090 24GB","priceHourText":"$0.35","ramGb":24,"vramGb":24},{"cc":"8.9","gpu":"RTX-4090","id":"gpu-rtx4090","label":"RTX 4090 24GB","priceHourText":"$0.54","ramGb":24,"vramGb":24},{"cc":"7.5","gpu":"T4","id":"gpu-t4","label":"T4 16GB","priceHourText":"$0.30","ramGb":16,"vramGb":16},{"cc":"12.0","gpu":"RTX-5090","id":"gpu-rtx5090","label":"RTX 5090 32GB","priceHourText":"$1.42","ramGb":32,"vramGb":32}],"outputKinds":["auto","image","video","audio","json","text","file","model3d"],"queueApi":{"ensureAndRun":"/api/cogs/run","join":"/api/cogs/{model_id}/queue/join","status":"/api/cogs/predictions/{prediction_id}"},"sessionsApi":{"close":"/api/cogs/sessions/{session_id}","create":"/api/cogs/{model_id}/sessions","status":"/api/cogs/sessions/{session_id}","ws":"/ws/cogs/sessions/{session_id}"},"spacesApi":{"accepts":["owner/name","https://huggingface.co/spaces/owner/name"],"import":"/api/cogs/spaces/import","port":7860,"runtime":"gradio"},"success":true,"templates":[{"name":"fastserve-qwen-27b","image":"ghcr.io/lee101/fastserve-qwen-3.8-27b:latest","hardware":"gpu-rtx5090","description":"Uncensored Qwen3.8-27B GGUF served by llama.cpp behind a native C front door. The 5090 profile uses Q4_K_M weights, Q8 KV cache, flash attention, prompt caching, and the retained MTP head for speculative decoding. Select dynamic-v3 to A/B the stock Unsloth Dynamic 3.0 UD-Q4_K_XL quant. The container answers both OpenAI-compatible /v1 requests and Cog predictions.","category":"Text","repository":"https://github.com/lee101/fastserve-qwen-3.8-27b","upstream":"https://huggingface.co/Qwen/Qwen3.8-27B","license":"MIT adapter · Apache-2.0 Qwen-derived weights · JonathanColetti and Unsloth model cards apply","tags":["llm","qwen3.8","llama.cpp","gguf","dynamic-3.0","uncensored","speculative-decoding","mtp","tool-calling","reasoning","openai-api"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"User message","required":true,"order":0},{"name":"system","type":"string","description":"Optional system prompt","required":false,"order":1},{"name":"image","type":"image","description":"Optional image — Qwen3.8-27B is image-text-to-text","required":false,"order":2},{"name":"max_tokens","type":"integer","description":"Maximum tokens to generate","default":1024,"required":false,"min":1,"max":32768,"order":3},{"name":"temperature","type":"number","description":"Sampling temperature","default":0.7,"required":false,"min":0,"max":2,"order":4},{"name":"top_p","type":"number","description":"Nucleus sampling","default":0.8,"required":false,"min":0,"max":1,"order":5}],"outputKind":"text"},"idleSeconds":20,"minVramGb":32,"diskGb":80,"modelVariants":["auto","uncensored-gguf","dynamic-v3"],"defaultModelVariant":"auto","priceHourText":"$1.42"},{"name":"anima-2.9b","image":"ghcr.io/lee101/anima-omniserve-native:latest","hardware":"gpu-l40s","description":"Anima-2.9B anime and illustration generation through a native Cog: BF16, TF32, SDPA, batched classifier-free guidance, a precompiled torch.compile cache so cold workers never compile on a request, persistent Hugging Face cache, and RunPod scale-to-zero.","category":"Image","repository":"https://github.com/lee101/anima-omniserve-native","upstream":"https://huggingface.co/Gazingstars123/Anima-2.9B","license":"CircleStone Labs commercial license","tags":["image","anime","illustration","cosmos","bf16","torch-compile","serverless"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Detailed anime or illustration prompt","required":true,"order":0},{"name":"negative_prompt","type":"string","description":"Optional negative prompt","required":false,"order":1},{"name":"width","type":"integer","description":"Width, snapped to a 16px grid","default":832,"required":false,"min":512,"max":1536,"order":2},{"name":"height","type":"integer","description":"Height, snapped to a 16px grid","default":1216,"required":false,"min":512,"max":1536,"order":3},{"name":"num_inference_steps","type":"integer","description":"Euler flow-matching steps","default":28,"required":false,"min":10,"max":50,"order":4},{"name":"guidance_scale","type":"number","description":"Classifier-free guidance","default":4,"required":false,"min":1,"max":8,"order":5},{"name":"seed","type":"integer","description":"Use -1 for a random seed","default":-1,"required":false,"min":-1,"max":2147483647,"order":6},{"name":"output_format","type":"string","default":"webp","required":false,"choices":["webp","png","jpeg"],"order":7}],"outputKind":"image"},"idleSeconds":180,"minVramGb":48,"diskGb":80,"serverless":true,"serverlessDockerArgs":"python -u /src/runpod_handler.py","priceHourText":"$1.65"},{"name":"qwen-image-flash-space","image":"registry.hf.space/akhaliq-qwen-image-flash:latest","hardware":"gpu-h100","description":"Run akhaliq/Qwen-Image-Flash's existing Hugging Face Space image unchanged on an app.nz H100. Gradio UI and named API are bridged into the same billed, scale-to-zero Cog lifecycle.","category":"Image","repository":"https://huggingface.co/spaces/akhaliq/Qwen-Image-Flash","upstream":"https://huggingface.co/nvidia/Qwen-Image-Flash","license":"NVIDIA Open Model License · Space source license unspecified","verified":true,"tags":["hugging-face","space","gradio","qwen-image","dmd2","4-step","text-to-image"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Image prompt","required":true,"order":0},{"name":"negative_prompt","type":"string","description":"Optional negative prompt","required":false,"order":1},{"name":"width","type":"integer","description":"Width (snapped to a multiple of 16)","default":1024,"required":false,"min":256,"max":2048,"order":2},{"name":"height","type":"integer","description":"Height (snapped to a multiple of 16)","default":1024,"required":false,"min":256,"max":2048,"order":3},{"name":"num_inference_steps","type":"integer","description":"DMD2 inference steps (4 is the distilled trajectory)","default":4,"required":false,"min":1,"max":12,"order":4},{"name":"seed","type":"integer","description":"Random seed","default":0,"required":false,"min":0,"max":2147483647,"order":5},{"name":"randomize_seed","type":"boolean","description":"Choose a new seed for every image","default":true,"required":false,"order":6}],"outputKind":"image","runtime":"gradio","apiName":"generate_image","spaceId":"akhaliq/Qwen-Image-Flash","dockerArgs":"python app.py"},"idleSeconds":180,"minVramGb":60,"diskGb":80,"priceHourText":"$4.78"},{"name":"qwen-image-flash-cog","image":"registry.app.nz/appnz-qwen-image-flash:latest","hardware":"gpu-h100","description":"Native app.nz Cog adapter for NVIDIA Qwen-Image-Flash. Four-step BF16 inference, persistent Hugging Face cache, generated prediction API/UI, and H100 scale-to-zero hosting.","category":"Image","repository":"https://github.com/lee101/app-site/tree/master/replicatecog/appnz-qwen-image-flash","upstream":"https://huggingface.co/nvidia/Qwen-Image-Flash","license":"Apache-2.0 adapter · NVIDIA Open Model License weights","verified":true,"tags":["cog","qwen-image","dmd2","4-step","h100","text-to-image"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Image prompt","required":true,"order":0},{"name":"negative_prompt","type":"string","description":"Optional negative prompt","required":false,"order":1},{"name":"width","type":"integer","default":1024,"required":false,"min":256,"max":2048,"order":2},{"name":"height","type":"integer","default":1024,"required":false,"min":256,"max":2048,"order":3},{"name":"num_inference_steps","type":"integer","default":4,"required":false,"min":1,"max":12,"order":4},{"name":"seed","type":"integer","description":"Use -1 for a random seed","default":-1,"required":false,"min":-1,"max":2147483647,"order":5},{"name":"output_format","type":"string","default":"webp","required":false,"choices":["webp","png","jpeg"],"order":6}],"outputKind":"image"},"idleSeconds":180,"minVramGb":60,"diskGb":80,"priceHourText":"$4.78"},{"name":"fast-vfx","image":"r8.im/lee101/fast-vfx","hardware":"gpu-rtx3090","description":"GPU video FX: color-band quantization with NVENC/NVDEC accelerated ffmpeg. Verified end-to-end on RunPod.","schema":{"inputs":[{"name":"video","type":"video","description":"Input video file","required":true,"order":0},{"name":"num_levels","type":"integer","description":"Number of color levels","default":25,"required":false,"min":2,"max":256,"order":1},{"name":"output_codec","type":"string","description":"Output codec","default":"h264_nvenc","required":false,"choices":["h264_nvenc","av1_nvenc"],"order":2}],"outputKind":"video"},"priceHourText":"$0.35"},{"name":"abot-world","image":"ghcr.io/lee101/appnz-abot-world:latest","hardware":"gpu-rtx4090","description":"ABot-World interactive world model (Alibaba AMAP, 0.5B): action-conditioned video simulation you can walk through with WASD over a session websocket. Source: github.com/lee101/appnz-abot-world.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Scene to imagine","required":false,"order":0},{"name":"keys","type":"string","description":"Held action keys (w/a/s/d) — predictions shim only","required":false,"order":1},{"name":"steps","type":"integer","description":"Frames to simulate before returning one","default":1,"required":false,"min":1,"max":64,"order":2}],"outputKind":"image","session":true},"idleSeconds":120,"minVramGb":20,"priceHourText":"$0.54"},{"name":"ardy","image":"ghcr.io/lee101/appnz-ardy:latest","hardware":"gpu-rtx4090","description":"ARDY real-time humanoid motion (NVIDIA, Apache-2.0): stream text prompts, waypoints, and WASD over a session websocket; JSON pose frames out at 20 FPS. Source: github.com/lee101/appnz-ardy.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Motion description (streamable mid-session)","required":false,"order":0},{"name":"steps","type":"integer","description":"Pose frames to generate (predictions shim)","default":40,"required":false,"min":1,"max":400,"order":1}],"outputKind":"json","session":true},"idleSeconds":120,"minVramGb":20,"priceHourText":"$0.54"},{"name":"appnz-tts","image":"ghcr.io/lee101/appnz-tts@sha256:172711eff82541fa33eb9bf3fc10cd6d0320d3f837e8ff99609b1a3c4e30b7dd","hardware":"gpu-rtx3090","description":"app.nz text-to-speech (fish-speech). Stored or inline reference voices, mp3/opus/wav out. Runs local on prod when the GPU has room, uses RunPod Serverless for low traffic, and promotes to a pod only under sustained load.","category":"Audio","repository":"https://github.com/lee101/app-site/tree/master/replicatecog/appnz-tts","upstream":"https://github.com/fishaudio/fish-speech","license":"Open adapter · upstream/model terms apply","tags":["tts","fish-speech","voice-cloning","serverless"],"schema":{"inputs":[{"name":"input","type":"string","description":"Text to synthesize","default":"Hi, let's think through the best things in the world.","required":true,"order":0},{"name":"voice","type":"string","description":"Stored voice id","required":false,"order":1},{"name":"voice_sample","type":"audio","description":"Inline reference sample (clone this voice)","required":false,"order":2},{"name":"response_format","type":"string","description":"Audio format","default":"mp3","required":false,"choices":["mp3","opus","wav"],"order":3},{"name":"speed","type":"number","description":"Playback speed","default":1,"required":false,"min":0.25,"max":4,"order":4},{"name":"language","type":"string","description":"Optional language hint","required":false,"order":5}],"outputKind":"audio"},"idleSeconds":30,"minVramGb":16,"serverless":true,"serverlessDockerArgs":"python -u /app/runpod_handler.py","priceHourText":"$0.42"},{"name":"pocket-tts","image":"ghcr.io/lee101/pocket-tts-cog:latest","hardware":"gpu-rtx3090","description":"Kyutai pocket-tts: 100M-parameter CPU-first TTS, 26 voices across 6 languages, and authorized voice cloning from a sample. Code is MIT; weights are CC-BY-4.0 and bundled voice licenses vary. Source: github.com/lee101/pocket-tts-cog.","category":"Audio","repository":"https://github.com/lee101/pocket-tts-cog","upstream":"https://github.com/kyutai-labs/pocket-tts","license":"MIT adapter/code · CC-BY-4.0 model weights · voice licenses vary","verified":true,"tags":["tts","cpu","streaming","voice-cloning","multilingual"],"schema":{"inputs":[{"name":"text","type":"string","description":"Text to speak","required":true,"order":0},{"name":"voice","type":"string","description":"Voice","default":"alba","required":false,"choices":["alba","anna","azelma","bill_boerst","caro_davy","charles","cosette","eponine","eve","fantine","george","jane","jean","javert","marius","mary","michael","paul","peter_yearsley","stuart_bell","vera","giovanni","lola","juergen","rafael","estelle"],"order":1},{"name":"voice_sample","type":"audio","description":"Optional reference audio to clone","required":false,"order":2},{"name":"format","type":"string","description":"Audio format","default":"mp3","required":false,"choices":["mp3","opus","wav"],"order":3}],"outputKind":"audio"},"idleSeconds":60,"priceHourText":"$0.35"},{"name":"piano-generator","image":"registry.app.nz/appnz-piano-generator:latest","hardware":"gpu-rtx3090","description":"Three reproducible CC0-trained piano models in one tiny Cog. Run the same image through app.nz Cog HTTP inference or a RunPod Serverless handler; stereo Opus output and CPU-friendly synthesis.","category":"Audio","repository":"https://github.com/lee101/app-site/tree/master/replicatecog/appnz-piano-generator","upstream":"https://github.com/lee101/app-site/tree/master/training/music","license":"MIT adapter and trainer · bundled seed corpus CC0-1.0","verified":true,"tags":["piano","music","open-model","runpod","cog","symbolic","opus","cpu-friendly"],"schema":{"inputs":[{"name":"model","type":"string","description":"Bundled open piano model","default":"open-piano-baroque","required":false,"choices":["open-piano-baroque","open-piano-jazz","open-piano-minimal"],"order":0},{"name":"sound","type":"string","description":"Piano synthesis character","default":"studio","required":false,"choices":["felt","studio","bright"],"order":1},{"name":"bars","type":"integer","description":"Song length in four-beat bars","default":8,"required":false,"min":1,"max":32,"order":2},{"name":"tempo","type":"integer","description":"Tempo in BPM","default":108,"required":false,"min":40,"max":220,"order":3},{"name":"key","type":"string","description":"Major key","default":"C","required":false,"choices":["C","D","E","F","G","A","B"],"order":4},{"name":"seed","type":"integer","description":"Deterministic seed","default":17,"required":false,"min":0,"max":2147483647,"order":5},{"name":"policy_url","type":"string","description":"Optional app.nz/R2 n-gram or RL policy artifact override","required":false,"order":6}],"outputKind":"audio"},"idleSeconds":30,"diskGb":2,"serverless":true,"serverlessDockerArgs":"python -u /app/runpod_handler.py","priceHourText":"$0.42"},{"name":"music-diffusion","image":"registry.app.nz/appnz-music-diffusion:latest","hardware":"gpu-a100","description":"ACE-Step 1.5 full-song diffusion Cog with lyrics, tempo, key, seed, and denoising controls. Checkpoints stay on the persistent model cache and outputs are native Opus.","category":"Audio","repository":"https://github.com/lee101/app-site/tree/master/replicatecog/appnz-music-diffusion","upstream":"https://github.com/ace-step/ACE-Step-1.5","license":"Apache-2.0 adapter/upstream code · checkpoint model card applies","tags":["music","diffusion","ace-step","lyrics","opus","lora"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Style, instrumentation, mood, and production prompt","required":true,"order":0},{"name":"lyrics","type":"string","description":"Structured lyrics or [Instrumental]","default":"[Instrumental]","required":false,"order":1},{"name":"duration","type":"integer","description":"Duration in seconds","default":30,"required":false,"min":10,"max":300,"order":2},{"name":"bpm","type":"integer","description":"Tempo in BPM","default":120,"required":false,"min":40,"max":240,"order":3},{"name":"key","type":"string","description":"Key and scale","default":"C Major","required":false,"order":4},{"name":"seed","type":"integer","description":"Deterministic seed","default":17,"required":false,"min":0,"max":2147483647,"order":5},{"name":"steps","type":"integer","description":"Turbo diffusion steps","default":8,"required":false,"min":1,"max":20,"order":6}],"outputKind":"audio"},"idleSeconds":180,"minVramGb":24,"diskGb":40,"priceHourText":"$2.62"},{"name":"yue2","image":"ghcr.io/lee101/yue-cog:latest","hardware":"gpu-rtx4090","description":"YuE2 lyrics-to-song: style + tagged lyrics in, 48kHz stereo song out. bf16 AR + flow-matching NAR with CUDA graphs, tiled VAE decode, persistent warm pipeline. Weights fetch at runtime (~7GB) into the volume cache.","category":"Audio","repository":"https://github.com/lee101/yue-cog","upstream":"https://github.com/multimodal-art-projection/YuE","license":"MIT adapter · YuE2 weights CC BY-NC 4.0 + creator permission for commercial use","tags":["music","song","lyrics-to-song","yue2","runpod","cog"],"schema":{"inputs":[{"name":"style","type":"string","description":"Genre, instruments, vocal character, language, tempo","required":true,"order":0},{"name":"lyrics","type":"string","description":"Lyrics with [Verse]/[Chorus] section tags","required":true,"order":1},{"name":"cot","type":"string","description":"Symbolic planning: full melody+chords, melody only, or off","default":"full","required":false,"choices":["full","melody","off"],"order":2},{"name":"seed","type":"integer","description":"Random seed","default":831001,"required":false,"min":0,"order":3},{"name":"abc","type":"string","description":"Optional own ABC score (requires full/melody)","required":false,"order":4},{"name":"cfg_scale","type":"number","description":"Text guidance, blank for default","default":0,"required":false,"min":0,"max":20,"order":5},{"name":"ode_steps","type":"integer","description":"Flow-matching steps; 32 quality default, 16 fast","default":32,"required":false,"min":1,"max":128,"order":6},{"name":"semantic_max_tokens","type":"integer","description":"Cap song length for tests; 9000 full song","default":9000,"required":false,"min":200,"max":9000,"order":7},{"name":"format","type":"string","description":"Delivery audio format","default":"mp3","required":false,"choices":["mp3","wav","flac"],"order":8}],"outputKind":"audio"},"idleSeconds":10,"minVramGb":24,"diskGb":40,"serverless":true,"serverlessDockerArgs":"python -u /src/rp_handler.py","priceHourText":"$0.65"},{"name":"pocket-tts-gpu","image":"ghcr.io/lee101/pocket-tts-cog:cuda","hardware":"gpu-rtx3090","description":"pocket-tts on CUDA: ~3.5x realtime on a modern GPU (vs ~1x/core on busy server CPUs). Same voices/cloning as pocket-tts. Source: github.com/lee101/pocket-tts-cog.","category":"Audio","repository":"https://github.com/lee101/pocket-tts-cog","upstream":"https://github.com/kyutai-labs/pocket-tts","license":"MIT adapter/code · CC-BY-4.0 model weights · voice licenses vary","verified":true,"tags":["tts","cuda","streaming","voice-cloning","multilingual"],"schema":{"inputs":[{"name":"text","type":"string","description":"Text to speak","required":true,"order":0},{"name":"voice","type":"string","description":"Voice","default":"alba","required":false,"choices":["alba","anna","azelma","bill_boerst","caro_davy","charles","cosette","eponine","eve","fantine","george","jane","jean","javert","marius","mary","michael","paul","peter_yearsley","stuart_bell","vera","giovanni","lola","juergen","rafael","estelle"],"order":1},{"name":"voice_sample","type":"audio","description":"Optional reference audio to clone","required":false,"order":2},{"name":"format","type":"string","description":"Audio format","default":"mp3","required":false,"choices":["mp3","opus","wav"],"order":3}],"outputKind":"audio"},"idleSeconds":60,"priceHourText":"$0.35"},{"name":"vocal-separator","image":"ghcr.io/lee101/appnz-vocal-separator:latest","hardware":"gpu-rtx3090","description":"Demucs htdemucs source separation: export vocals + instrumental or a ZIP of all four stems. CUDA 12.8 build supports Ampere through Blackwell.","category":"Audio","repository":"https://github.com/lee101/appnz-vocal-separator","upstream":"https://github.com/facebookresearch/demucs","license":"MIT","verified":true,"tags":["demucs","music","stems","blackwell"],"schema":{"inputs":[{"name":"audio","type":"audio","description":"Input track (https URL or upload)","required":true,"order":0},{"name":"stems","type":"string","description":"Stem split","default":"two","required":false,"choices":["two","four"],"order":1},{"name":"format","type":"string","description":"Stem audio format","default":"mp3","required":false,"choices":["mp3","wav"],"order":2},{"name":"model","type":"string","description":"Demucs checkpoint","default":"htdemucs","required":false,"choices":["htdemucs","htdemucs_ft","mdx_extra"],"order":3}],"outputKind":"file"},"idleSeconds":120,"minVramGb":8,"priceHourText":"$0.35"},{"name":"image-upscaler","image":"ghcr.io/lee101/appnz-image-upscaler:latest","hardware":"gpu-rtx3090","description":"Real-ESRGAN 2×/4× restoration with optional GFPGAN face enhancement. Tiled CUDA 12.8 inference supports Ampere through Blackwell.","category":"Image","repository":"https://github.com/lee101/appnz-image-upscaler","upstream":"https://github.com/xinntao/Real-ESRGAN","license":"MIT adapter · BSD-3-Clause upstream","verified":true,"tags":["real-esrgan","gfpgan","upscale","blackwell"],"schema":{"inputs":[{"name":"image","type":"image","description":"Input image (https URL or upload)","required":true,"order":0},{"name":"scale","type":"integer","description":"Upscale factor (2 or 4)","default":4,"required":false,"choices":["2","4"],"order":1},{"name":"model","type":"string","description":"Restoration model","default":"general","required":false,"choices":["general","anime"],"order":2},{"name":"face_enhance","type":"boolean","description":"GFPGAN face enhancement","default":false,"required":false,"order":3}],"outputKind":"image"},"idleSeconds":90,"minVramGb":8,"priceHourText":"$0.35"},{"name":"background-remover","image":"ghcr.io/lee101/appnz-rembg:latest","hardware":"gpu-rtx3090","description":"rembg background removal to transparent PNG. The default IS-Net session is baked in and reused; alternate models load once on demand.","category":"Image","repository":"https://github.com/lee101/appnz-rembg","upstream":"https://github.com/danielgatis/rembg","license":"MIT","verified":true,"tags":["rembg","segmentation","transparent-png","cpu-friendly"],"schema":{"inputs":[{"name":"image","type":"image","description":"Input image (https URL or upload)","required":true,"order":0},{"name":"model","type":"string","description":"Segmentation model","default":"isnet-general-use","required":false,"choices":["isnet-general-use","u2net","u2netp","isnet-anime","birefnet-general"],"order":1}],"outputKind":"image"},"idleSeconds":60,"priceHourText":"$0.35"},{"name":"depth-vfx","image":"ghcr.io/lee101/appnz-depth-vfx:latest","hardware":"gpu-rtx3090","description":"Depth Anything V2 Small generates depth, heatmap, and normal passes or a subtle looping parallax MP4. The pinned 25M-parameter model stays warm and runs FP16/channels-last.","category":"Image","repository":"https://github.com/lee101/appnz-depth-vfx","upstream":"https://huggingface.co/depth-anything/Depth-Anything-V2-Small-hf","license":"MIT adapter · Apache-2.0 model","verified":true,"tags":["depth-anything-v2","parallax","normal-map","vfx"],"schema":{"inputs":[{"name":"image","type":"image","description":"Source image","required":true,"order":0},{"name":"effect","type":"string","description":"Depth-powered output","default":"parallax","required":false,"choices":["parallax","depth","heatmap","normals"],"order":1},{"name":"strength","type":"number","description":"Parallax displacement","default":0.55,"required":false,"min":0.1,"max":1,"order":2},{"name":"seconds","type":"integer","description":"Parallax duration","default":3,"required":false,"min":1,"max":6,"order":3}],"outputKind":"file"},"idleSeconds":60,"minVramGb":4,"priceHourText":"$0.35"},{"name":"lama-cleaner","image":"ghcr.io/lee101/appnz-lama-cleaner:latest","hardware":"gpu-rtx3090","description":"Prompt-free masked object removal with a checksum-pinned LaMa ONNX model. CUDA/CPU provider fallback, minimal padding, and feathered blending preserve pixels outside the repair.","category":"Image","repository":"https://github.com/lee101/appnz-lama-cleaner","upstream":"https://github.com/advimman/lama","license":"MIT adapter · Apache-2.0 model","verified":true,"tags":["lama","inpainting","object-removal","onnx"],"schema":{"inputs":[{"name":"image","type":"image","description":"Source image","required":true,"order":0},{"name":"mask","type":"image","description":"White areas are removed","required":true,"order":1},{"name":"grow","type":"integer","description":"Expand the removal mask","default":8,"required":false,"min":0,"max":64,"order":2},{"name":"feather","type":"integer","description":"Blend the repaired edge","default":6,"required":false,"min":0,"max":32,"order":3}],"outputKind":"image"},"idleSeconds":60,"minVramGb":4,"priceHourText":"$0.35"},{"name":"voxel-3d","image":"ghcr.io/lee101/appnz-voxel-3d:latest","hardware":"gpu-rtx3090","description":"A combined image-to-3D pipeline: U2NetP isolation, TripoSR reconstruction, marching cubes, optional color-preserving voxelization, and browser-ready GLB export.","category":"3D","repository":"https://github.com/lee101/appnz-voxel-3d","upstream":"https://github.com/VAST-AI-Research/TripoSR","license":"MIT","verified":true,"tags":["triposr","image-to-3d","voxel","glb"],"schema":{"inputs":[{"name":"image","type":"image","description":"Single object on a simple background","required":true,"order":0},{"name":"style","type":"string","description":"Mesh or voxelized GLB","default":"voxel48","required":false,"choices":["mesh","voxel32","voxel48","voxel64"],"order":1},{"name":"foreground_ratio","type":"number","description":"Object size in the normalized view","default":0.85,"required":false,"min":0.5,"max":0.95,"order":2},{"name":"mc_resolution","type":"integer","description":"Surface extraction resolution","default":256,"required":false,"choices":["128","192","256"],"order":3}],"outputKind":"file"},"idleSeconds":90,"minVramGb":8,"diskGb":20,"priceHourText":"$0.35"},{"name":"pixal3d","image":"ghcr.io/lee101/pixal3dcog:latest","hardware":"gpu-rtx4090","description":"Pixal3D (SIGGRAPH 2026, TRELLIS.2 backbone): single image to a textured PBR GLB with pixel-aligned detail. Automatic matting and camera estimation, 1024 or 1536 cascade, remeshed and decimated output with baked base color, metallic and roughness. Source: github.com/lee101/pixal3dcog.","category":"3D","repository":"https://github.com/lee101/pixal3dcog","upstream":"https://github.com/TencentARC/Pixal3D","license":"MIT","verified":true,"tags":["pixal3d","trellis2","image-to-3d","pbr","glb"],"schema":{"inputs":[{"name":"image","type":"image","description":"Reference image; alpha is used as the mask when present","required":true,"order":0},{"name":"resolution","type":"integer","description":"Cascade resolution (1024 is ~2x faster)","default":1536,"required":false,"choices":["1024","1536"],"order":1},{"name":"texture_size","type":"integer","description":"Baked texture size","default":2048,"required":false,"choices":["1024","2048","4096"],"order":2},{"name":"decimation_target","type":"integer","description":"Target face count","default":500000,"required":false,"min":10000,"max":2000000,"order":3},{"name":"remesh","type":"boolean","description":"Remesh before baking","default":true,"required":false,"order":4},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":5},{"name":"fov","type":"number","description":"Manual horizontal FOV in radians (blank = MoGe-2 estimate)","required":false,"order":6}],"outputKind":"model3d"},"idleSeconds":120,"minVramGb":24,"diskGb":40,"priceHourText":"$0.54"},{"name":"vectorizer","image":"ghcr.io/lee101/appnz-vectorizer:latest","hardware":"gpu-rtx3090","description":"Deterministic raster-to-SVG conversion using VTracer's multithreaded Rust core. No model weights or GPU runtime; optimized for fast cold starts and compact paths.","category":"Image","repository":"https://github.com/lee101/appnz-vectorizer","upstream":"https://github.com/visioncortex/vtracer","license":"MIT","verified":true,"tags":["svg","vector","vtracer","cpu"],"schema":{"inputs":[{"name":"image","type":"image","description":"Raster artwork","required":true,"order":0},{"name":"preset","type":"string","description":"Tracing style","default":"poster","required":false,"choices":["poster","logo","photo","pixel"],"order":1},{"name":"speckle","type":"integer","description":"Discard regions smaller than this","default":6,"required":false,"min":0,"max":64,"order":2},{"name":"color_precision","type":"integer","description":"Bits retained per color channel","default":6,"required":false,"min":3,"max":8,"order":3}],"outputKind":"file"},"idleSeconds":30,"priceHourText":"$0.35"},{"name":"audex-s2s","image":"ghcr.io/lee101/appnz-audex-s2s:latest","hardware":"gpu-rtx4090","description":"Speech-to-speech conversation: NVIDIA Nemotron-Labs-Audex-2B (unified audio-text LLM) hears you, reasons, and talks back — ASR + chat + TTS in one 2B model on vLLM. Powers /spaces/audio-to-audio. Noncommercial weights pulled from HF at boot. Open source: app.nz/repos/replicatecog/audex.","schema":{"inputs":[{"name":"task","type":"string","description":"converse: speech in, spoken reply out. transcribe / speak / chat for single stages.","default":"converse","required":false,"choices":["converse","transcribe","speak","chat"],"order":0},{"name":"audio","type":"audio","description":"Your spoken turn (wav/mp3/ogg/webm/m4a)","required":false,"order":1},{"name":"text","type":"string","description":"Text input for speak/chat; for converse it overrides ASR","required":false,"order":2},{"name":"system_prompt","type":"string","description":"Character or instructions for the reply voice","required":false,"order":3},{"name":"history","type":"string","description":"Prior turns as JSON: [{\"role\":\"user\",\"text\":\"hi\"},{\"role\":\"assistant\",\"text\":\"hey!\"}]","required":false,"order":4},{"name":"enable_reasoning","type":"boolean","description":"Think before replying (better answers, slower)","default":false,"required":false,"order":5},{"name":"audio_format","type":"string","description":"Reply audio format","default":"mp3","required":false,"choices":["mp3","wav"],"order":6}],"outputKind":"json"},"idleSeconds":300,"minVramGb":20,"priceHourText":"$0.54"},{"name":"media-encode","image":"appnz/media-encode:latest","hardware":"gpu-t4","description":"ffmpeg media encoding: transcode mp4/webm, extract mp3, gif, thumbnails, waveforms. No weights — boots in seconds on the cheapest tier.","schema":{"inputs":[{"name":"media","type":"file","description":"Input video/audio (https URL or upload)","required":true,"order":0},{"name":"operation","type":"string","description":"Encode operation","default":"transcode-mp4","required":true,"choices":["audio-mp3","gif","thumbnail","transcode-mp4","transcode-webm","waveform"],"order":1},{"name":"width","type":"integer","description":"Scale output to this width","required":false,"min":16,"max":4096,"order":2},{"name":"start","type":"number","description":"Trim start (seconds)","required":false,"order":3},{"name":"duration","type":"number","description":"Trim duration (seconds)","required":false,"order":4}],"outputKind":"file"},"idleSeconds":120,"priceHourText":"$0.30"},{"name":"media-optimizer","image":"appnz/media-optimizer:latest","hardware":"gpu-rtx4090","description":"Cloudflare-style image resizing plus AV1 video ladders. Mount /artifacts, emit WebP/AVIF/JPEG image variants and AV1/WebM or MP4 video renditions.","schema":{"inputs":[{"name":"media","type":"file","description":"Input image/video file, HTTPS URL, or mounted artifact path","required":true,"order":0},{"name":"operation","type":"string","description":"Optimizer operation","default":"image-responsive","required":true,"choices":["image-responsive","image-cdn-auto","video-av1-ladder"],"order":1},{"name":"widths","type":"string","description":"Comma-separated image widths, e.g. 320,640,1280","default":"320,640,960,1280","required":false,"order":2},{"name":"heights","type":"string","description":"Comma-separated video heights, e.g. 360,720,1080","default":"360,720,1080","required":false,"order":3},{"name":"formats","type":"string","description":"Comma-separated output formats: webp,avif,jpeg or webm,mp4","default":"webp","required":false,"order":4},{"name":"quality","type":"integer","description":"Visual quality target","default":85,"required":false,"min":1,"max":100,"order":5}],"outputKind":"json"},"idleSeconds":90,"priceHourText":"$0.54"},{"name":"embeddings","image":"appnz/embeddings:latest","hardware":"gpu-t4","description":"Sentence embeddings (all-MiniLM-L6-v2). One text per line in, JSON vectors out — pairs with notebooks and datasets for search/dedup/clustering.","schema":{"inputs":[{"name":"texts","type":"string","description":"Texts to embed, one per line (or a JSON array)","required":true,"order":0},{"name":"normalize","type":"boolean","description":"L2-normalize vectors","default":true,"required":false,"order":1}],"outputKind":"json"},"idleSeconds":120,"priceHourText":"$0.30"},{"name":"wan22-t2v","image":"ghcr.io/lee101/accelerated-cgtaylor-wan-app-nz:t2v","hardware":"gpu-l40s","description":"Wan2.2 A14B text-to-video, CG-Taylor accelerated (~1.3x, training-free) + NF4. Weights pulled from the app.nz R2 model mirror. Source: github.com/lee101/accelerated-cgtaylor-wan-app-nz.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Text prompt","required":true,"order":0},{"name":"negative_prompt","type":"string","description":"Negative prompt","default":"","required":false,"order":1},{"name":"num_frames","type":"integer","description":"Frames (16fps)","default":81,"required":false,"min":17,"max":121,"order":2},{"name":"resolution","type":"string","description":"WxH","default":"480x480","required":false,"choices":["480x480","480x832","832x480","720x1280","1280x720"],"order":3},{"name":"steps","type":"integer","description":"Denoise steps","default":27,"required":false,"min":8,"max":50,"order":4},{"name":"guidance","type":"number","description":"Guidance scale","default":4,"required":false,"min":1,"max":10,"order":5},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":6},{"name":"cgtaylor","type":"boolean","description":"CG-Taylor caching accel","default":true,"required":false,"order":7},{"name":"image","type":"image","description":"Ignored for t2v; the container requires the field","default":"","required":false,"order":8}],"outputKind":"video"},"idleSeconds":300,"minVramGb":40,"diskGb":200,"priceHourText":"$1.38"},{"name":"wan22-i2v","image":"ghcr.io/lee101/accelerated-cgtaylor-wan-app-nz:i2v","hardware":"gpu-l40s","description":"Wan2.2 A14B image-to-video (NF4, auto resolution from input aspect). Weights from the app.nz R2 model mirror. Source: github.com/lee101/accelerated-cgtaylor-wan-app-nz.","schema":{"inputs":[{"name":"image","type":"image","description":"Input image","required":true,"order":0},{"name":"prompt","type":"string","description":"Motion/content prompt","required":true,"order":1},{"name":"negative_prompt","type":"string","description":"Negative prompt","default":"","required":false,"order":2},{"name":"num_frames","type":"integer","description":"Frames (16fps)","default":81,"required":false,"min":17,"max":121,"order":3},{"name":"steps","type":"integer","description":"Denoise steps","default":27,"required":false,"min":8,"max":50,"order":4},{"name":"guidance","type":"number","description":"Guidance scale","default":3.5,"required":false,"min":1,"max":10,"order":5},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":6},{"name":"cgtaylor","type":"boolean","description":"CG-Taylor caching accel","default":true,"required":false,"order":7}],"outputKind":"video"},"idleSeconds":300,"minVramGb":40,"diskGb":200,"priceHourText":"$1.38"},{"name":"wan-animate-2","image":"ghcr.io/lee101/wan-animate-cog@sha256:c6607ae517f87e9dbb49e2c0d0ea1595111dcf9e14e9d0455caa1ad2b97f12f3","hardware":"gpu-l40s","description":"Wan-Animate-2 14B distilled character animation from a reference image and raw driving video. Quality-first Euler inference, row-wise dynamic FP8 on Ada 48GB GPUs, driving-audio preservation, and automatic serverless-to-pod routing.","category":"Video","repository":"https://github.com/lee101/wan-animate-cog","upstream":"https://github.com/Wan-Video/Wan-Animate-2","license":"Apache-2.0 adapter, code, and model weights; input-media rights remain with the user","verified":true,"tags":["video","character-animation","dance","raw-driving-video","wan-animate-2","fp8","serverless"],"schema":{"inputs":[{"name":"image","type":"image","description":"Reference character image","default":"https://appstatic.app.nz/static/cogs/wan-animate-2/reference.png","required":true,"order":0},{"name":"driving_video","type":"video","description":"Raw motion/expression video; optional audio is preserved","default":"https://appstatic.app.nz/static/cogs/wan-animate-2/driving.mp4","required":true,"order":1},{"name":"prompt","type":"string","description":"Objective character appearance and background caption","default":"A full-body character dancing naturally, stable identity and clothing, continuous motion.","required":true,"order":2},{"name":"quality","type":"string","description":"Output pixel-area tier; reference aspect ratio is preserved","default":"preview","required":false,"choices":["preview","balanced","high"],"order":3},{"name":"max_seconds","type":"number","description":"Trim the driving video to this duration","default":1,"required":false,"min":1,"max":15,"order":4},{"name":"fps","type":"integer","description":"Motion sampling and output frame rate","default":12,"required":false,"choices":["12","16","24","30"],"order":5},{"name":"frames_per_segment","type":"integer","description":"Must be 4n+1","default":17,"required":false,"min":17,"max":81,"order":6},{"name":"steps","type":"integer","description":"Distilled denoise steps","default":6,"required":false,"min":6,"max":20,"order":7},{"name":"seed","type":"integer","description":"Random seed","default":43,"required":false,"order":8},{"name":"preserve_audio","type":"boolean","description":"Mux driving-video audio into the generated MP4","default":false,"required":false,"order":9},{"name":"cgtaylor","type":"boolean","description":"Experimental conservative denoiser prediction","default":false,"required":false,"order":10},{"name":"cgtaylor_threshold","type":"number","description":"Maximum relative calibration error before one predicted denoiser call","default":0.015,"required":false,"min":0.001,"max":0.05,"order":11}],"outputKind":"video"},"idleSeconds":30,"minVramGb":48,"diskGb":100,"serverless":true,"serverlessDockerArgs":"python -u /src/rp_handler.py","priceHourText":"$1.65"},{"name":"liveavatar","image":"ghcr.io/lee101/liveavatar-app-nz:appnz","hardware":"gpu-h100","description":"LiveAvatar v1.1 audio-driven character animation: reference image + speech or singing + scene prompt to a long-form MP4. Four-step DMD inference, FP8, persistent warm pipeline. Apache-2.0 source: github.com/lee101/liveavatar-app-nz.","schema":{"inputs":[{"name":"image","type":"image","description":"Reference portrait or character image","required":true,"order":0},{"name":"audio","type":"audio","description":"Driving speech or singing (max 300 seconds)","required":true,"order":1},{"name":"prompt","type":"string","description":"Appearance, action, camera, and scene guidance","default":"A character speaks naturally with expressive facial movement and body gestures.","required":false,"order":2},{"name":"quality","type":"string","description":"Pixel-area tier; input aspect ratio is preserved","default":"standard","required":false,"choices":["preview","standard","high"],"order":3},{"name":"num_clips","type":"integer","description":"Maximum autoregressive clips; also stops when audio ends","default":1,"required":false,"min":1,"max":8,"order":4},{"name":"steps","type":"integer","description":"DMD sampling steps; 4 is recommended","default":4,"required":false,"min":4,"max":8,"order":5},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":6},{"name":"start_from_reference","type":"boolean","description":"Use the reference image as frame zero","default":true,"required":false,"order":7}],"outputKind":"video"},"idleSeconds":300,"minVramGb":80,"diskGb":80,"priceHourText":"$4.78"},{"name":"splat-capture","image":"ghcr.io/lee101/appnz-splat:latest","hardware":"gpu-rtx4090","description":"Photos to gaussian splat: VGGT-Commercial pose+point init, gsplat refine, optional 2DGS+TSDF textured mesh. Exports .splat/.ply/.glb. Source: splat-worker/ in app-site.","schema":{"inputs":[{"name":"images","type":"string","description":"Image URLs, one per line (2-64 photos)","required":true,"order":0},{"name":"tier","type":"string","description":"Quality tier","default":"fast","required":false,"choices":["fast","quality"],"order":1},{"name":"mesh","type":"boolean","description":"Extract textured GLB mesh","default":false,"required":false,"order":2},{"name":"max_splats","type":"integer","description":"Splat budget","default":1500000,"required":false,"min":100000,"max":4000000,"order":3},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":4}],"outputKind":"model3d"},"idleSeconds":120,"minVramGb":24,"priceHourText":"$0.54"},{"name":"ltx23-fast","image":"ghcr.io/lee101/appnz-ltx-fast:latest","hardware":"gpu-rtx5090","description":"LTX-2.3 22B distilled 8-step FP8 + SageAttention: image/text-to-video with native audio, up to 4K via spatial upscaler. ~4 GPU-s per video-second at 720p. Source: ltx-worker/ in app-site.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Prompt","required":true,"order":0},{"name":"image","type":"image","description":"First frame (image-to-video when set)","required":false,"order":1},{"name":"negative_prompt","type":"string","description":"Negative prompt","required":false,"order":2},{"name":"resolution","type":"string","description":"Output resolution","default":"1080p","required":false,"choices":["720p","1080p","1440p","2160p"],"order":3},{"name":"seconds","type":"integer","description":"Duration","default":5,"required":false,"min":1,"max":20,"order":4},{"name":"fps","type":"integer","description":"Frames per second","default":24,"required":false,"choices":["24","48"],"order":5},{"name":"audio","type":"boolean","description":"Generate synced audio","default":false,"required":false,"order":6},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":7}],"outputKind":"video"},"idleSeconds":180,"minVramGb":32,"priceHourText":"$1.42"},{"name":"minimax-h3","image":"ghcr.io/lee101/h3-cog:e8b6929","hardware":"gpu-rtx5090","description":"MiniMax H3 text/image-to-video with native stereo audio, optional uploaded driving audio (Ref2VA), first/last-frame control, true loop conditioning, R2/HF weight cache, and GPU AV1 output. Set APPNZ_MINIMAX_H3_LICENSE_ACCEPTED=1 to clear the launch gate; prefers local NZ Docker when VRAM allows.","category":"Video","repository":"https://github.com/lee101/h3-cog","upstream":"https://huggingface.co/MiniMaxAI/MiniMax-H3","license":"MiniMax Model License (territory-restricted; adapter MIT)","tags":["video","text-to-video","image-to-video","first-last-frame","audio-driven","audio","av1","loop","serverless"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Scene, camera, action, style, and optional audio direction. For audio drive, mention speaking to camera.","required":true,"order":0},{"name":"first_frame","type":"image","description":"Optional first frame for I2V, loop, or audio-driven identity (required with audio)","required":false,"order":1},{"name":"last_frame","type":"image","description":"Optional target final frame (not used with audio drive)","required":false,"order":2},{"name":"audio","type":"audio","description":"Optional driving speech/performance (2–15s). Requires first_frame; mouth timing follows this clip via Ref2VA","required":false,"order":3},{"name":"aspect_ratio","type":"string","description":"Output aspect ratio","default":"16:9","required":false,"choices":["16:9","9:16","1:1","4:3","3:4","21:9"],"order":4},{"name":"size","type":"string","description":"Preview, balanced, or native 768px short-edge canvas","default":"balanced","required":false,"choices":["preview","balanced","native"],"order":5},{"name":"duration","type":"number","description":"Duration in seconds at 24 fps (ignored when audio is set — uses audio length)","default":5,"required":false,"min":4,"max":15,"order":6},{"name":"steps","type":"integer","description":"Sampling steps; 20 is the quality default","default":20,"required":false,"min":8,"max":30,"order":7},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":8},{"name":"structured_prompt","type":"boolean","description":"Expand concise text into H3's structured prompt format","default":true,"required":false,"order":9},{"name":"loop","type":"boolean","description":"Reuse first_frame as the final-frame condition for a seamless loop","default":false,"required":false,"order":10},{"name":"include_audio","type":"boolean","description":"Keep H3 native stereo audio in the delivered video","default":true,"required":false,"order":11},{"name":"output_codec","type":"string","description":"Delivery codec","default":"webm-av1","required":false,"choices":["webm-av1","mp4-h264"],"order":12},{"name":"encode_quality","type":"integer","description":"NVENC constant quality (lower is higher quality)","default":26,"required":false,"min":16,"max":45,"order":13}],"outputKind":"video"},"idleSeconds":180,"minVramGb":32,"diskGb":96,"serverless":true,"serverlessDockerArgs":"python -u /src/rp_handler.py","platformFeePercent":80,"priceHourText":"$1.92"},{"name":"yue2-music","image":"ghcr.io/lee101/yue-cog:latest","hardware":"gpu-rtx4090","description":"YuE2 lyrics-to-song: editable melody-and-chord planning rendered as 48kHz stereo with vocals. BF16 quality path (no fp16), CUDA-graph backend, persistent warm pipeline.","category":"Audio","repository":"https://github.com/lee101/yue-cog","upstream":"https://github.com/multimodal-art-projection/YuE","license":"MIT adapter · YuE2 weights CC BY-NC 4.0 (commercial use needs author license)","tags":["music","song","lyrics","vocals","symbolic-planning","serverless"],"schema":{"inputs":[{"name":"style","type":"string","description":"Genre, instruments, vocal character, language, tempo","required":true,"order":0},{"name":"lyrics","type":"string","description":"Lyrics with [Verse]/[Chorus] section tags","required":true,"order":1},{"name":"cot","type":"string","description":"Symbolic planning: full chords+melody, melody only, or off","default":"full","required":false,"choices":["full","melody","off"],"order":2},{"name":"seed","type":"integer","description":"Random seed","default":831001,"required":false,"order":3},{"name":"ode_steps","type":"integer","description":"Flow-matching steps; 32 quality default, 16 fast","default":32,"required":false,"choices":["16","24","32","48"],"order":4},{"name":"semantic_max_tokens","type":"integer","description":"Cap song length for cheap tests; 9000 full song","default":9000,"required":false,"choices":["1500","3000","6000","9000"],"order":5},{"name":"format","type":"string","description":"Delivery audio format","default":"mp3","required":false,"choices":["mp3","wav","flac"],"order":6}],"outputKind":"audio"},"idleSeconds":60,"minVramGb":24,"diskGb":40,"serverless":true,"serverlessDockerArgs":"python -u /src/rp_handler.py","priceHourText":"$0.65"},{"name":"ltxv23-lora","image":"ghcr.io/lee101/ltxv23-lora-app-nz:latest","hardware":"gpu-h100","description":"LTX-2.3 22b distilled text/image-to-video with synced audio and hot-swappable LoRAs (cinemagraph, camera controls, styles — or any .safetensors URL). Caching LoRA server included. Source: github.com/lee101/ltxv23-lora-app-nz.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Text prompt (also drives generated audio)","required":true,"order":0},{"name":"image","type":"image","description":"Optional first-frame image (image-to-video)","required":false,"order":1},{"name":"lora","type":"string","description":"Preset LoRA (trigger words auto-prepended)","default":"none","required":false,"choices":["none","cinemagraph","transition","realism","real-human","camera-controls","motion-stabilizer","pixel-art","pop","googly-eyes","fast-paced","equirectangular","panoramic-360","sci-fi-cinema","product-ad","claymation","paper-cutout","fantasy-anime","beauty-of-rain","marbling","xianxia"],"order":2},{"name":"lora_url","type":"string","description":"Custom LoRA .safetensors URL (overrides preset)","required":false,"order":3},{"name":"lora_strength","type":"number","description":"LoRA strength","default":1,"required":false,"min":0,"max":2,"order":4},{"name":"num_frames","type":"integer","description":"Frames (snapped to 8k+1; 121 ≈ 5s at 24fps)","default":121,"required":false,"min":9,"max":257,"order":5},{"name":"resolution","type":"string","description":"WxH (multiples of 64) or auto","default":"auto","required":false,"choices":["auto","1280x704","704x1280","960x960","1920x1088","1088x1920"],"order":6},{"name":"frame_rate","type":"number","description":"FPS","default":24,"required":false,"min":8,"max":50,"order":7},{"name":"enhance_prompt","type":"boolean","description":"LLM-expand the prompt","default":false,"required":false,"order":8},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":9}],"outputKind":"video"},"idleSeconds":300,"minVramGb":80,"diskGb":120,"priceHourText":"$4.78"},{"name":"alaya-world","image":"ghcr.io/lee101/appnz-alayaworld:latest","hardware":"gpu-h100","description":"AlayaWorld interactive autoregressive world model (AlayaLab): first frame + camera path + prompt -\u003e minute-long, camera-steerable video with long-horizon memory. Source-available under the LTX-2 Community License (free under $10M ARR) — not OSI open source; see the repo's THIRD_PARTY.md. Source: github.com/lee101/appnz-alayaworld.","schema":{"inputs":[{"name":"image","type":"image","description":"First frame that seeds the clip","required":true,"order":0},{"name":"prompt","type":"string","description":"Text prompt for the whole clip","required":true,"order":1},{"name":"camera","type":"string","description":"Camera path preset","default":"dolly_forward","required":false,"choices":["static","dolly_forward","dolly_backward","strafe_left","strafe_right","rise","descend","pan_left","pan_right","custom"],"order":2},{"name":"camera_pt","type":"file","description":"Custom trajectory .pt (cam_c2w [F,4,4] + optional intrinsic [3,3]); used when camera=custom","required":false,"order":3},{"name":"duration_seconds","type":"number","description":"Clip length in seconds","default":8,"required":false,"min":1.3,"max":60,"order":4},{"name":"skill_prompt","type":"string","description":"One-off end effect: swap the prompt for the final skill_seconds","required":false,"order":5},{"name":"skill_seconds","type":"number","description":"Duration of the end-skill effect","default":4,"required":false,"min":0,"max":10,"order":6},{"name":"ttc","type":"boolean","description":"Pathwise Test-Time Correction (curbs drift on long clips)","default":false,"required":false,"order":7},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":8},{"name":"video_crf","type":"integer","description":"h264 quality: 18 near-lossless, 28 small","default":28,"required":false,"min":18,"max":32,"order":9}],"outputKind":"video"},"idleSeconds":300,"minVramGb":80,"diskGb":140,"priceHourText":"$4.78"},{"name":"sdxl-turbo","image":"r8.im/stability-ai/sdxl-turbo","hardware":"gpu-rtx4090","description":"Fast text-to-image. GATED on r8.im — add an r8.im registry credential first, or build your own image on Build Studio.","schema":{"inputs":[{"name":"prompt","type":"string","description":"Text prompt","required":true,"order":0},{"name":"width","type":"integer","default":512,"required":false,"min":256,"max":1024,"order":1},{"name":"height","type":"integer","default":512,"required":false,"min":256,"max":1024,"order":2},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":3}],"outputKind":"image"},"priceHourText":"$0.54"},{"name":"omniserve","image":"ghcr.io/lee101/omniserve:latest","hardware":"gpu-rtx4090","description":"The omni model server: diffusion + video + LoRA + LLM in one cog. Auto-loads models on demand, VRAM-balances with sleep/evict tiers, OpenAI-compatible when run as a server. LTX video needs H100. Open source: github.com/lee101/omniserve.","schema":{"inputs":[{"name":"task","type":"string","description":"Inference task","default":"image","required":true,"choices":["image","video","chat"],"order":0},{"name":"model","type":"string","description":"Catalog model key, or auto for the family default","default":"auto","required":false,"choices":["auto","z-image-turbo","flux-schnell","sdxl-turbo","qwen-image","ltx-2.3-distilled","qwen3-4b-instruct","qwen3-32b","gemma-3-12b-it"],"order":1},{"name":"prompt","type":"string","description":"Prompt (or chat user message)","required":true,"order":2},{"name":"system_prompt","type":"string","description":"System prompt for chat","required":false,"order":3},{"name":"negative_prompt","type":"string","description":"Negative prompt (image)","required":false,"order":4},{"name":"width","type":"integer","default":1024,"required":false,"min":256,"max":2048,"order":5},{"name":"height","type":"integer","default":1024,"required":false,"min":256,"max":2048,"order":6},{"name":"steps","type":"integer","description":"0 = model default","default":0,"required":false,"min":0,"max":100,"order":7},{"name":"num_frames","type":"integer","description":"Video frames (snapped to 8k+1)","default":121,"required":false,"min":9,"max":257,"order":8},{"name":"lora","type":"string","description":"LoRA catalog id or https url","required":false,"order":9},{"name":"lora_strength","type":"number","default":1,"required":false,"min":0,"max":2,"order":10},{"name":"max_tokens","type":"integer","description":"Chat max tokens","default":1024,"required":false,"min":1,"max":32768,"order":11},{"name":"temperature","type":"number","default":0.7,"required":false,"min":0,"max":2,"order":12},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":13}],"outputKind":"auto"},"idleSeconds":300,"minVramGb":24,"priceHourText":"$0.54"},{"name":"qwen-image-2.1","image":"ghcr.io/lee101/omniserve-native:ra2","hardware":"gpu-rtx4090","description":"Qwen Image 2.1 (Q4_K_M DiT + Qwen3VL-8B text encoder, stable-diffusion.cpp CUDA) text-to-image and reference edit, WebP/PNG. Serverless-first: worker 0 at rest, promoted to a pod only when sustained traffic is cheaper.","category":"Image","tags":["image","text-to-image","image-edit","qwen","qwen-image-2.1","ra2","serverless"],"schema":{"inputs":[{"name":"prompt","type":"string","description":"Text prompt","required":true,"order":0},{"name":"negative_prompt","type":"string","description":"Negative prompt","default":"","required":false,"order":1},{"name":"width","type":"integer","description":"Output width in pixels","default":1024,"required":false,"order":2},{"name":"height","type":"integer","description":"Output height in pixels","default":1024,"required":false,"order":3},{"name":"steps","type":"integer","description":"Sampling steps","default":20,"required":false,"order":4},{"name":"guidance_scale","type":"number","description":"Classifier-free guidance scale","default":1,"required":false,"order":5},{"name":"seed","type":"integer","description":"Random seed (blank = random)","required":false,"order":6},{"name":"image_base64","type":"string","description":"Base64 PNG/JPEG reference image; set to edit that image instead of text-to-image","required":false,"order":7},{"name":"strength","type":"number","description":"Reference-edit denoise strength","default":0.6,"required":false,"order":8},{"name":"output_format","type":"string","description":"Output image format","default":"webp","required":false,"choices":["webp","png","jpeg"],"order":9},{"name":"cache","type":"boolean","description":"Reuse a cached result for an identical request","default":false,"required":false,"order":10}],"outputKind":"image","dockerArgs":"python -u /opt/omniserve/runtime/cog_pod.py"},"idleSeconds":120,"minVramGb":16,"diskGb":40,"serverless":true,"serverlessDockerArgs":"python -u /opt/omniserve/runtime/handler.py","priceHourText":"$0.73"}]}
