view qwen3_vl/runtime.bzl @ 275:78699f810817

Add Qwen3-VL and WebRTC dictation services Add Bazel targets for the CUDA-backed Qwen3-VL server and a local WebRTC faster-whisper dictation service. Co-authored-by: Copilot <[email protected]> Copilot-Session: e3d8cb06-6c95-4ae0-9757-651d3796ab00
author MrJuneJune <me@mrjunejune.com>
date Mon, 17 Aug 2026 10:58:47 -0700
parents
children
line wrap: on
line source

def _cuda_runtime_impl(ctx):
    output = ctx.actions.declare_directory(ctx.label.name)
    inputs = ctx.files.llama_runtime + ctx.files.cuda_runtime
    args = ctx.actions.args()
    args.add(output.path)
    args.add_all([file.path for file in inputs])

    ctx.actions.run_shell(
        inputs = inputs,
        outputs = [output],
        arguments = [args],
        command = """
set -euo pipefail
output="$1"
shift
mkdir -p "$output"
for input in "$@"; do
  cp -L "$input" "$output/$(basename "$input")"
done
chmod 0555 "$output"/*.exe
chmod 0444 "$output"/*.dll
""",
        progress_message = "Assembling llama.cpp CUDA runtime",
    )

    return [
        DefaultInfo(
            files = depset([output]),
            runfiles = ctx.runfiles(files = [output]),
        ),
    ]

cuda_runtime = rule(
    implementation = _cuda_runtime_impl,
    attrs = {
        "llama_runtime": attr.label(
            allow_files = True,
            mandatory = True,
        ),
        "cuda_runtime": attr.label(
            allow_files = True,
            mandatory = True,
        ),
    },
)