feat: 增加 NVIDIA 动态库依赖并默认自动选择计算类型

This commit is contained in:
cat-shark
2026-08-13 22:01:42 +08:00
parent 206275f7d6
commit 812d276998
4 changed files with 60 additions and 1 deletions
+2
View File
@@ -5,6 +5,8 @@ description = "WOV faster-whisper ASR node"
requires-python = ">=3.11"
dependencies = [
"faster-whisper>=1.2.1",
"nvidia-cublas-cu12>=12.9.2.10",
"nvidia-cudnn-cu12>=9.24.0.43",
"wov-sdk",
]
+17
View File
@@ -15,7 +15,10 @@ class FakeSegment:
class FakeWhisperModel:
instances: list[tuple[tuple, dict]] = []
def __init__(self, *args, **kwargs):
FakeWhisperModel.instances.append((args, kwargs))
self.args = args
self.kwargs = kwargs
@@ -54,6 +57,7 @@ def test_format_timestamp() -> None:
def test_success(tmp_path, monkeypatch) -> None:
FakeWhisperModel.instances.clear()
_install_fake_whisper(monkeypatch)
(tmp_path / "audio.wav").write_bytes(b"fake")
response = invoke(_request(tmp_path))
@@ -61,6 +65,19 @@ def test_success(tmp_path, monkeypatch) -> None:
content = Path(response.outputs["srt_uri"]).read_text(encoding="utf-8")
assert "第一段" in content
assert "01:00:00,500 --> 01:00:02,250" in content
_, kwargs = FakeWhisperModel.instances[-1]
assert kwargs["device"] == "auto"
assert kwargs["compute_type"] == "auto"
def test_compute_type_override(tmp_path, monkeypatch) -> None:
FakeWhisperModel.instances.clear()
_install_fake_whisper(monkeypatch)
(tmp_path / "audio.wav").write_bytes(b"fake")
response = invoke(_request(tmp_path, params={"language": "ja", "compute_type": "int8"}))
assert response.status == "completed"
_, kwargs = FakeWhisperModel.instances[-1]
assert kwargs["compute_type"] == "int8"
def test_model_raises(tmp_path, monkeypatch) -> None:
Generated
+40
View File
@@ -509,6 +509,42 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/a1/5a/4d2b1601df3602dba7a14f3348ba9bfe94a18adb428e693df6154c293831/numpy-2.5.1-cp314-cp314t-win_arm64.whl", hash = "sha256:5a6db61f9aaa57e369905c67d852045d3c4f7126405b29d09b19dec118e9c9cb", size = 10697674, upload-time = "2026-07-04T17:07:58.506Z" },
]
[[package]]
name = "nvidia-cublas-cu12"
version = "12.9.2.10"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "nvidia-cuda-nvrtc-cu12" },
]
wheels = [
{ url = "https://files.pythonhosted.org/packages/f7/a2/c96163a0fff1839c0c9548bbdeae7b853b867009e33b9b9264adc238b1cf/nvidia_cublas_cu12-12.9.2.10-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:5572131a59c3eebeeb1c4c8144f772d49372c20124916e072a0e3fc30df421d5", size = 575012079, upload-time = "2026-04-08T18:51:47.303Z" },
{ url = "https://files.pythonhosted.org/packages/cb/c0/0a517bfe63ccd3b92eb254d264e28fca3c7cab75d07daea315250fb1bf73/nvidia_cublas_cu12-12.9.2.10-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:e4f53a8ca8c5d6e8c492d0d0a3d565ecb59a751b19cfdaa4f6da0ab2104c1702", size = 581240110, upload-time = "2026-04-08T18:52:31.532Z" },
{ url = "https://files.pythonhosted.org/packages/20/e2/fc9a0e985249d873150276d5afb02e39a66817fedbf1a385724393e505ed/nvidia_cublas_cu12-12.9.2.10-py3-none-win_amd64.whl", hash = "sha256:623f43027d40d44ceadf0043f002bd25cf353e8f13ce90b9a87057019f560661", size = 553162896, upload-time = "2026-04-08T18:53:10.035Z" },
]
[[package]]
name = "nvidia-cuda-nvrtc-cu12"
version = "12.9.86"
source = { registry = "https://pypi.org/simple" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/b8/85/e4af82cc9202023862090bfca4ea827d533329e925c758f0cde964cb54b7/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:210cf05005a447e29214e9ce50851e83fc5f4358df8b453155d5e1918094dcb4", size = 89568129, upload-time = "2025-06-05T20:02:41.973Z" },
{ url = "https://files.pythonhosted.org/packages/64/eb/c2295044b8f3b3b08860e2f6a912b702fc92568a167259df5dddb78f325e/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:096d4de6bda726415dfaf3198d4f5c522b8e70139c97feef5cd2ca6d4cd9cead", size = 44528905, upload-time = "2025-06-05T20:02:29.754Z" },
{ url = "https://files.pythonhosted.org/packages/52/de/823919be3b9d0ccbf1f784035423c5f18f4267fb0123558d58b813c6ec86/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-win_amd64.whl", hash = "sha256:72972ebdcf504d69462d3bcd67e7b81edd25d0fb85a2c46d3ea3517666636349", size = 76408187, upload-time = "2025-06-05T20:12:27.819Z" },
]
[[package]]
name = "nvidia-cudnn-cu12"
version = "9.24.0.43"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "nvidia-cublas-cu12" },
]
wheels = [
{ url = "https://files.pythonhosted.org/packages/9c/f1/cd42563325fa827f54ff30da05686c747652bdbd4cb5654cea54d7d0ad4f/nvidia_cudnn_cu12-9.24.0.43-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:a42996943f0cd78ddfd61c8bf59361672a19b63e0491aa22a53d6fe63a3f854a", size = 856490582, upload-time = "2026-07-02T16:21:30.924Z" },
{ url = "https://files.pythonhosted.org/packages/10/13/b8887c869cf2471339a24b60d3c28e761facbb534935f572b61423371abb/nvidia_cudnn_cu12-9.24.0.43-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:f424192dd85e7d29f44be18df2dae4c80d32c67a29c0d42f5c283c40cfdf871c", size = 799083985, upload-time = "2026-07-02T16:25:37.467Z" },
{ url = "https://files.pythonhosted.org/packages/29/28/2c9a2a97a8b3fedcf74a14f38fd5edfae12274380a829fdc6b16ce29be4c/nvidia_cudnn_cu12-9.24.0.43-py3-none-win_amd64.whl", hash = "sha256:cbd41a0ab084422c936dc9fb2fc89be5ea9a85bc421c6f23d0243bdfc945fbef", size = 737103728, upload-time = "2026-07-02T16:30:10.901Z" },
]
[[package]]
name = "onnxruntime"
version = "1.28.0"
@@ -791,6 +827,8 @@ version = "0.1.0"
source = { virtual = "." }
dependencies = [
{ name = "faster-whisper" },
{ name = "nvidia-cublas-cu12" },
{ name = "nvidia-cudnn-cu12" },
{ name = "wov-sdk" },
]
@@ -803,6 +841,8 @@ dev = [
[package.metadata]
requires-dist = [
{ name = "faster-whisper", specifier = ">=1.2.1" },
{ name = "nvidia-cublas-cu12", specifier = ">=12.9.2.10" },
{ name = "nvidia-cudnn-cu12", specifier = ">=9.24.0.43" },
{ name = "wov-sdk", directory = "../wov-sdk" },
]
+1 -1
View File
@@ -33,7 +33,7 @@ def invoke(request: InvokeRequest) -> InvokeResponse:
or os.getenv("WHISPER_MODEL_PATH", "large-v3")
)
device = str(request.params.get("device") or os.getenv("WHISPER_DEVICE", "auto"))
compute_type = str(request.params.get("compute_type") or "float16")
compute_type = str(request.params.get("compute_type") or "auto")
model = WhisperModel(
model_path,
device=device,