diff --git a/pyproject.toml b/pyproject.toml index 5e83cb0..ccab6d8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,6 +5,8 @@ description = "WOV faster-whisper ASR node" requires-python = ">=3.11" dependencies = [ "faster-whisper>=1.2.1", + "nvidia-cublas-cu12>=12.9.2.10", + "nvidia-cudnn-cu12>=9.24.0.43", "wov-sdk", ] diff --git a/tests/test_whisper_node.py b/tests/test_whisper_node.py index 2eee36d..34f4b9f 100644 --- a/tests/test_whisper_node.py +++ b/tests/test_whisper_node.py @@ -15,7 +15,10 @@ class FakeSegment: class FakeWhisperModel: + instances: list[tuple[tuple, dict]] = [] + def __init__(self, *args, **kwargs): + FakeWhisperModel.instances.append((args, kwargs)) self.args = args self.kwargs = kwargs @@ -54,6 +57,7 @@ def test_format_timestamp() -> None: def test_success(tmp_path, monkeypatch) -> None: + FakeWhisperModel.instances.clear() _install_fake_whisper(monkeypatch) (tmp_path / "audio.wav").write_bytes(b"fake") response = invoke(_request(tmp_path)) @@ -61,6 +65,19 @@ def test_success(tmp_path, monkeypatch) -> None: content = Path(response.outputs["srt_uri"]).read_text(encoding="utf-8") assert "第一段" in content assert "01:00:00,500 --> 01:00:02,250" in content + _, kwargs = FakeWhisperModel.instances[-1] + assert kwargs["device"] == "auto" + assert kwargs["compute_type"] == "auto" + + +def test_compute_type_override(tmp_path, monkeypatch) -> None: + FakeWhisperModel.instances.clear() + _install_fake_whisper(monkeypatch) + (tmp_path / "audio.wav").write_bytes(b"fake") + response = invoke(_request(tmp_path, params={"language": "ja", "compute_type": "int8"})) + assert response.status == "completed" + _, kwargs = FakeWhisperModel.instances[-1] + assert kwargs["compute_type"] == "int8" def test_model_raises(tmp_path, monkeypatch) -> None: diff --git a/uv.lock b/uv.lock index dee2caf..9dd391f 100644 --- a/uv.lock +++ b/uv.lock @@ -509,6 +509,42 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a1/5a/4d2b1601df3602dba7a14f3348ba9bfe94a18adb428e693df6154c293831/numpy-2.5.1-cp314-cp314t-win_arm64.whl", hash = "sha256:5a6db61f9aaa57e369905c67d852045d3c4f7126405b29d09b19dec118e9c9cb", size = 10697674, upload-time = "2026-07-04T17:07:58.506Z" }, ] +[[package]] +name = "nvidia-cublas-cu12" +version = "12.9.2.10" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-cuda-nvrtc-cu12" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/f7/a2/c96163a0fff1839c0c9548bbdeae7b853b867009e33b9b9264adc238b1cf/nvidia_cublas_cu12-12.9.2.10-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:5572131a59c3eebeeb1c4c8144f772d49372c20124916e072a0e3fc30df421d5", size = 575012079, upload-time = "2026-04-08T18:51:47.303Z" }, + { url = "https://files.pythonhosted.org/packages/cb/c0/0a517bfe63ccd3b92eb254d264e28fca3c7cab75d07daea315250fb1bf73/nvidia_cublas_cu12-12.9.2.10-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:e4f53a8ca8c5d6e8c492d0d0a3d565ecb59a751b19cfdaa4f6da0ab2104c1702", size = 581240110, upload-time = "2026-04-08T18:52:31.532Z" }, + { url = "https://files.pythonhosted.org/packages/20/e2/fc9a0e985249d873150276d5afb02e39a66817fedbf1a385724393e505ed/nvidia_cublas_cu12-12.9.2.10-py3-none-win_amd64.whl", hash = "sha256:623f43027d40d44ceadf0043f002bd25cf353e8f13ce90b9a87057019f560661", size = 553162896, upload-time = "2026-04-08T18:53:10.035Z" }, +] + +[[package]] +name = "nvidia-cuda-nvrtc-cu12" +version = "12.9.86" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b8/85/e4af82cc9202023862090bfca4ea827d533329e925c758f0cde964cb54b7/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-manylinux2010_x86_64.manylinux_2_12_x86_64.whl", hash = "sha256:210cf05005a447e29214e9ce50851e83fc5f4358df8b453155d5e1918094dcb4", size = 89568129, upload-time = "2025-06-05T20:02:41.973Z" }, + { url = "https://files.pythonhosted.org/packages/64/eb/c2295044b8f3b3b08860e2f6a912b702fc92568a167259df5dddb78f325e/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:096d4de6bda726415dfaf3198d4f5c522b8e70139c97feef5cd2ca6d4cd9cead", size = 44528905, upload-time = "2025-06-05T20:02:29.754Z" }, + { url = "https://files.pythonhosted.org/packages/52/de/823919be3b9d0ccbf1f784035423c5f18f4267fb0123558d58b813c6ec86/nvidia_cuda_nvrtc_cu12-12.9.86-py3-none-win_amd64.whl", hash = "sha256:72972ebdcf504d69462d3bcd67e7b81edd25d0fb85a2c46d3ea3517666636349", size = 76408187, upload-time = "2025-06-05T20:12:27.819Z" }, +] + +[[package]] +name = "nvidia-cudnn-cu12" +version = "9.24.0.43" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "nvidia-cublas-cu12" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/9c/f1/cd42563325fa827f54ff30da05686c747652bdbd4cb5654cea54d7d0ad4f/nvidia_cudnn_cu12-9.24.0.43-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:a42996943f0cd78ddfd61c8bf59361672a19b63e0491aa22a53d6fe63a3f854a", size = 856490582, upload-time = "2026-07-02T16:21:30.924Z" }, + { url = "https://files.pythonhosted.org/packages/10/13/b8887c869cf2471339a24b60d3c28e761facbb534935f572b61423371abb/nvidia_cudnn_cu12-9.24.0.43-py3-none-manylinux_2_27_x86_64.whl", hash = "sha256:f424192dd85e7d29f44be18df2dae4c80d32c67a29c0d42f5c283c40cfdf871c", size = 799083985, upload-time = "2026-07-02T16:25:37.467Z" }, + { url = "https://files.pythonhosted.org/packages/29/28/2c9a2a97a8b3fedcf74a14f38fd5edfae12274380a829fdc6b16ce29be4c/nvidia_cudnn_cu12-9.24.0.43-py3-none-win_amd64.whl", hash = "sha256:cbd41a0ab084422c936dc9fb2fc89be5ea9a85bc421c6f23d0243bdfc945fbef", size = 737103728, upload-time = "2026-07-02T16:30:10.901Z" }, +] + [[package]] name = "onnxruntime" version = "1.28.0" @@ -791,6 +827,8 @@ version = "0.1.0" source = { virtual = "." } dependencies = [ { name = "faster-whisper" }, + { name = "nvidia-cublas-cu12" }, + { name = "nvidia-cudnn-cu12" }, { name = "wov-sdk" }, ] @@ -803,6 +841,8 @@ dev = [ [package.metadata] requires-dist = [ { name = "faster-whisper", specifier = ">=1.2.1" }, + { name = "nvidia-cublas-cu12", specifier = ">=12.9.2.10" }, + { name = "nvidia-cudnn-cu12", specifier = ">=9.24.0.43" }, { name = "wov-sdk", directory = "../wov-sdk" }, ] diff --git a/wov_node_whisper/__main__.py b/wov_node_whisper/__main__.py index 418a5e7..29f140f 100644 --- a/wov_node_whisper/__main__.py +++ b/wov_node_whisper/__main__.py @@ -33,7 +33,7 @@ def invoke(request: InvokeRequest) -> InvokeResponse: or os.getenv("WHISPER_MODEL_PATH", "large-v3") ) device = str(request.params.get("device") or os.getenv("WHISPER_DEVICE", "auto")) - compute_type = str(request.params.get("compute_type") or "float16") + compute_type = str(request.params.get("compute_type") or "auto") model = WhisperModel( model_path, device=device,