diff qwen3_vl/runtime.bzl @ 275:78699f810817

Add Qwen3-VL and WebRTC dictation services Add Bazel targets for the CUDA-backed Qwen3-VL server and a local WebRTC faster-whisper dictation service. Co-authored-by: Copilot <[email protected]> Copilot-Session: e3d8cb06-6c95-4ae0-9757-651d3796ab00
author MrJuneJune <me@mrjunejune.com>
date Mon, 17 Aug 2026 10:58:47 -0700
parents
children
line wrap: on
line diff
--- /dev/null	Thu Jan 01 00:00:00 1970 +0000
+++ b/qwen3_vl/runtime.bzl	Mon Aug 17 10:58:47 2026 -0700
@@ -0,0 +1,45 @@
+def _cuda_runtime_impl(ctx):
+    output = ctx.actions.declare_directory(ctx.label.name)
+    inputs = ctx.files.llama_runtime + ctx.files.cuda_runtime
+    args = ctx.actions.args()
+    args.add(output.path)
+    args.add_all([file.path for file in inputs])
+
+    ctx.actions.run_shell(
+        inputs = inputs,
+        outputs = [output],
+        arguments = [args],
+        command = """
+set -euo pipefail
+output="$1"
+shift
+mkdir -p "$output"
+for input in "$@"; do
+  cp -L "$input" "$output/$(basename "$input")"
+done
+chmod 0555 "$output"/*.exe
+chmod 0444 "$output"/*.dll
+""",
+        progress_message = "Assembling llama.cpp CUDA runtime",
+    )
+
+    return [
+        DefaultInfo(
+            files = depset([output]),
+            runfiles = ctx.runfiles(files = [output]),
+        ),
+    ]
+
+cuda_runtime = rule(
+    implementation = _cuda_runtime_impl,
+    attrs = {
+        "llama_runtime": attr.label(
+            allow_files = True,
+            mandatory = True,
+        ),
+        "cuda_runtime": attr.label(
+            allow_files = True,
+            mandatory = True,
+        ),
+    },
+)