Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 32 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -1158,6 +1158,9 @@ add_library(vllm STATIC
# Parakeet head archs so config.json RESOLVES (SupportsTranscription mirror).
src/vllm/multimodal/parakeet_transcription.cpp
src/vllm/model_executor/models/parakeet_registry.cpp
# Diarization seam (wraps parakeet.cpp's C-API). Gated by
# VLLM_CPP_WITH_DIARIZATION; the .cpp compiles as stubs when disabled.
src/vllm/multimodal/diarization.cpp
# The ONE video-generation seam every consumer drives (C ABI vllm_video_*,
# the server's /v1/videos, the minimax-h3-gen example) — ARCH-ONE-SURFACE
# ROW 2: absorbs the assembly pipeline examples/minimax_h3_gen and the
Expand Down Expand Up @@ -1568,6 +1571,35 @@ target_compile_definitions(blake3_vendored PRIVATE
target_include_directories(blake3_vendored PUBLIC third_party/blake3)
set_target_properties(blake3_vendored PROPERTIES POSITION_INDEPENDENT_CODE ON)
target_link_libraries(vllm PUBLIC blake3_vendored)

# --- Parakeet.cpp dependency (diarization + SAS) -----------------------------
# parakeet.cpp provides the Nemotron-3-Diarization Sortformer encoder, speaker
# head, streaming diarization (AOSC), and speaker-attributed ASR (SAS) merge.
# vllm.cpp's own ParakeetTranscriber handles ASR; this adds the diarization
# side through parakeet.cpp's C-API (parakeet_capi.h).
option(VLLM_CPP_WITH_DIARIZATION "Enable diarization support via parakeet.cpp" ON)
if(VLLM_CPP_WITH_DIARIZATION)
set(VLLM_CPP_PARAKEET_CPP_DIR "" CACHE PATH
"Path to a parakeet.cpp source tree. If empty, FetchContent from GitHub.")
if(VLLM_CPP_PARAKEET_CPP_DIR)
set(parakeet_cpp_SOURCE_DIR "${VLLM_CPP_PARAKEET_CPP_DIR}")
else()
include(FetchContent)
FetchContent_Declare(parakeet_cpp
GIT_REPOSITORY https://github.com/mudler/parakeet.cpp.git
GIT_TAG main
GIT_SHALLOW ON)
# parakeet.cpp builds its own ggml; we only need libparakeet.a and the
# headers. Suppress its tests/examples to keep the build lean.
set(PARAKEET_BUILD_TESTS OFF CACHE BOOL "" FORCE)
set(PARAKEET_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
FetchContent_MakeAvailable(parakeet_cpp)
endif()
target_include_directories(vllm PUBLIC
${parakeet_cpp_SOURCE_DIR}/include)
target_link_libraries(vllm PUBLIC parakeet)
target_compile_definitions(vllm PUBLIC VLLM_WITH_DIARIZATION=1)
endif()
if(WIN32)
target_link_libraries(vllm PUBLIC ws2_32)
endif()
Expand Down
79 changes: 78 additions & 1 deletion include/vllm.h
Original file line number Diff line number Diff line change
Expand Up @@ -369,7 +369,7 @@ extern "C" {
* KevModel/LayaModel runs the decision forward, CuaS1Forms runs the score
* forward. Non-matching architectures are refused by name. Every existing
* struct and call is byte-identical. */
#define VLLM_ABI_VERSION 29
#define VLLM_ABI_VERSION 30

/* ── Export macro ─────────────────────────────────────────────────────────────
* Marks the symbols that make up the stable ABI. Default visibility now; Task 3
Expand Down Expand Up @@ -1096,6 +1096,83 @@ VLLM_API vllm_status vllm_transcribe(vllm_engine* engine,
VLLM_API void vllm_transcription_free(vllm_transcription* out);


/* ── Speaker diarization (ABI v30) ───────────────────────────────────────────
* When the library is built with VLLM_CPP_WITH_DIARIZATION=ON (the default),
* a second engine handle can be loaded from a Nemotron-3-Diarization GGUF
* file. The diarization engine identifies who spoke when in a mono 16 kHz
* audio stream. It is independent of the ASR (Parakeet) engine — the two
* can be combined via vllm_transcribe_and_diarize.
*
* When VLLM_CPP_WITH_DIARIZATION=OFF, every function below returns
* VLLM_ERR_INVALID_ARGUMENT with a "not compiled in" message. */

/* One speaker segment. */
typedef struct vllm_speaker_segment {
int32_t speaker; /* 0-indexed speaker ID */
float start; /* seconds from audio start */
float end;
} vllm_speaker_segment;

/* Diarization result. OWNERSHIP: free with vllm_diarization_free. */
typedef struct vllm_diarization {
vllm_speaker_segment* segments;
int32_t n_segments;
} vllm_diarization;

/* Load a diarization GGUF file. Returns NULL on error
* (vllm_last_error carries the detail). */
VLLM_API vllm_engine* vllm_diarization_load(const char* gguf_path);

/* Diarize a WAV file. Returns VLLM_OK on success. */
VLLM_API vllm_status vllm_diarize_path(vllm_engine* diar_engine,
const char* wav_path,
vllm_diarization* out);

/* Diarize raw PCM (mono float32, 16 kHz). */
VLLM_API vllm_status vllm_diarize_pcm(vllm_engine* diar_engine,
const float* pcm, int64_t n_samples,
int32_t sample_rate,
vllm_diarization* out);

/* Free a diarization result. NULL is a no-op. */
VLLM_API void vllm_diarization_free(vllm_diarization* out);


/* ── Speaker-attributed ASR (ABI v30) ───────────────────────────────────────
* Combined transcription + diarization: runs both models on the same audio
* and merges word timestamps with speaker segments. The ASR engine must be
* a Parakeet checkpoint; the diarization engine must be a GGUF loaded with
* vllm_diarization_load. */

typedef struct vllm_speaker_utterance {
int32_t speaker;
char* text;
float start;
float end;
float conf;
} vllm_speaker_utterance;

typedef struct vllm_sas_result {
vllm_speaker_utterance* utterances;
int32_t n_utterances;
} vllm_sas_result;

/* Run combined ASR + diarization on a WAV file. */
VLLM_API vllm_status vllm_transcribe_and_diarize(
vllm_engine* asr_engine, vllm_engine* diar_engine,
const char* wav_path,
vllm_sas_result* out);

/* Run combined ASR + diarization on raw PCM. */
VLLM_API vllm_status vllm_transcribe_and_diarize_pcm(
vllm_engine* asr_engine, vllm_engine* diar_engine,
const float* pcm, int64_t n_samples, int32_t sample_rate,
vllm_sas_result* out);

/* Free a SAS result. Each utterance's .text is freed, then the array. */
VLLM_API void vllm_sas_result_free(vllm_sas_result* out);


/* ── Embeddings (ABI v15) ─────────────────────────────────────────────────────
* The embeddings/pooling slice of the ONE-SURFACE fold: an engine loaded from
* a POOLING (embedding) checkpoint — config.json architectures resolving to a
Expand Down
40 changes: 40 additions & 0 deletions include/vllm/entrypoints/openai/api_server.h
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,9 @@
#include "vllm/entrypoints/openai/video_api.h"
#include "vllm/entrypoints/openai/speech_api.h"
#include "vllm/multimodal/parakeet_transcription.h"
#ifdef VLLM_WITH_DIARIZATION
#include "vllm/multimodal/diarization.h"
#endif

namespace vllm::tok {
class Tokenizer;
Expand Down Expand Up @@ -140,6 +143,12 @@ class ApiServer {
// extracts the upload and calls this with the raw file bytes.
DispatchResult handle_audio_transcriptions(
const std::string& file_bytes, const std::string& response_format) const;
#ifdef VLLM_WITH_DIARIZATION
DispatchResult handle_audio_diarizations(
const std::string& file_bytes, const std::string& response_format) const;
DispatchResult handle_audio_sas(
const std::string& file_bytes, const std::string& response_format) const;
#endif

// POST /v1/embeddings (ARCH-ONE-SURFACE ROW 6). Mirror of vLLM's
// pooling/embed/api_router.py:28 `create_embedding` over the
Expand Down Expand Up @@ -267,6 +276,27 @@ class ApiServer {
transcriber_ = std::move(transcriber);
}

#ifdef VLLM_WITH_DIARIZATION
// Attach the diarization seam backing POST /v1/audio/diarizations (ABI v30).
// ADDITIVE and OPT-IN: absent => route unregistered => 404, byte-identical to
// a server without diarization. The callback wraps the parakeet.cpp C-API.
using DiarizeFn =
std::function<std::vector<vllm::multimodal::SpeakerSegment>(
const uint8_t* wav_bytes, size_t num_bytes)>;
void set_diarizer(DiarizeFn diarizer) {
diarizer_ = std::move(diarizer);
}

// Attach the SAS seam backing POST /v1/audio/sas (speaker-attributed ASR).
// Runs both ASR and diarization on the same audio and merges the results.
using SasFn =
std::function<vllm::multimodal::SpeakerAttributedASR(
const uint8_t* wav_bytes, size_t num_bytes)>;
void set_sas(SasFn sas) {
sas_ = std::move(sas);
}
#endif

// Attach the embedding seam backing POST /v1/embeddings (ARCH-ONE-SURFACE
// ROW 6). ADDITIVE and OPT-IN like the transcriber above: absent => route
// unregistered => 404, byte-identical to a server without pooling. The
Expand Down Expand Up @@ -402,6 +432,12 @@ class ApiServer {
// or zero when the diagnostic legacy-dynamic mode is selected.
size_t http_worker_count() const;

// Expose the diarizer and SAS callbacks for the route handlers.
#ifdef VLLM_WITH_DIARIZATION
DiarizeFn diarizer_callback() const { return diarizer_; }
SasFn sas_callback() const { return sas_; }
#endif

private:
// Null in the serving-less (transcription-only) construction: the generate
// routes are then not registered, and direct handler dispatch reports the
Expand All @@ -415,6 +451,10 @@ class ApiServer {
const v1::metrics::PrometheusStatLogger* metrics_ = nullptr;
::vllm::openai::VideoRunner video_runner_;
TranscribeFn transcriber_;
#ifdef VLLM_WITH_DIARIZATION
DiarizeFn diarizer_;
SasFn sas_;
#endif
EmbedFn embedder_;
NerFn ner_;
ScoreFn score_;
Expand Down
90 changes: 90 additions & 0 deletions include/vllm/multimodal/diarization.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
// diarization.h — diarization seam wrapping parakeet.cpp's C-API
//
// VLLM_WITH_DIARIZATION gates the whole seam. When the parakeet.cpp
// dependency is absent (VLLM_CPP_WITH_DIARIZATION=OFF), the header
// is empty and every function is a no-op stub, so the rest of vllm.cpp
// compiles unchanged.
#pragma once

#include <memory>
#include <string>
#include <vector>
#include <cstdint>

#ifdef VLLM_WITH_DIARIZATION
#include "parakeet_capi.h"
#endif

namespace vllm::multimodal {

// One speaker segment: who spoke, and when.
struct SpeakerSegment {
int speaker; // 0-indexed speaker ID
float start; // seconds from audio start
float end;
};

// One speaker-attributed utterance.
struct SpeakerUtterance {
int speaker;
std::string text;
float start;
float end;
float conf;
};

// A loaded diarization model (Nemotron-3-Diarization GGUF).
// Wraps parakeet_ctx from parakeet.cpp's C-API.
class Diarizer {
public:
// Load a diarization GGUF file. Returns nullptr on failure.
static std::unique_ptr<Diarizer> FromFile(const std::string& path);

~Diarizer();

// Diarize a mono float32 waveform at `sample_rate`.
std::vector<SpeakerSegment> Diarize(
const float* pcm, int64_t n_samples, int sample_rate) const;

// Diarize a WAV file path.
std::vector<SpeakerSegment> DiarizeWavFile(const std::string& path) const;

#ifdef VLLM_WITH_DIARIZATION
// Access the underlying parakeet_ctx for streaming + SAS composition.
parakeet_ctx* ctx() const { return ctx_; }
#endif

private:
Diarizer();
#ifdef VLLM_WITH_DIARIZATION
parakeet_ctx* ctx_ = nullptr;
#endif
};

// Combined ASR + diarization: runs both models and merges the results.
// Takes an ASR engine (from vllm_engine_load with a Parakeet checkpoint)
// and a diarization model. The ASR path goes through the existing
// ParakeetTranscriber; the diarization path goes through parakeet.cpp's
// C-API. The merge uses parakeet.cpp's SAS merge layer.
struct SpeakerAttributedASR {
// The speaker-attributed utterances.
std::vector<SpeakerUtterance> utterances;
// True if at least one utterance was produced.
bool has_result = false;
};

// Run combined ASR + diarization on a WAV file.
// `asr_dir` is a Parakeet checkpoint directory (HF format).
// `diar_gguf` is a diarization GGUF file path.
SpeakerAttributedASR TranscribeAndDiarize(
const std::string& wav_path,
const std::string& asr_dir,
const std::string& diar_gguf);

// Run combined ASR + diarization on raw PCM.
SpeakerAttributedASR TranscribeAndDiarizePCM(
const float* pcm, int64_t n_samples, int sample_rate,
const std::string& asr_dir,
const std::string& diar_gguf);

} // namespace vllm::multimodal
Loading
Loading