Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
30 commits
Select commit Hold shift + click to select a range
f0e22dd
feat(parakeet-cpp): load diarization and CED models and companions
mudler Sep 28, 2026
699c17b
fix(parakeet-cpp): reset role fields on a failed companion load
mudler Sep 28, 2026
30b4394
feat(parakeet-cpp): add speaker diarization
mudler Sep 28, 2026
4eb8ed0
feat(parakeet-cpp): add sound event detection
mudler Sep 28, 2026
4a21997
fix(parakeet-cpp): cancel sound detection mid-feed, shrink the lock
mudler Sep 28, 2026
dc53ef6
feat(parakeet-cpp): stream speaker and sound events during live trans…
mudler Sep 28, 2026
f156c5e
fix(parakeet-cpp): keep scene events off the ASR critical path in live
mudler Sep 28, 2026
bacaf14
feat(realtime): surface live speaker and sound events
mudler Sep 28, 2026
694b410
fix(realtime): keep start/end on a zero-second transcription segment
mudler Sep 28, 2026
cfef2c8
chore(gallery): add parakeet-cpp diarization, CED and realtime scene …
mudler Sep 28, 2026
57497db
docs: document parakeet-cpp diarization, sound detection and live sce…
mudler Sep 28, 2026
5354cb4
fix(gallery): correct the realtime-scene license and wording nits
mudler Sep 28, 2026
9e66c71
fix(parakeet-cpp): reject a companion role that duplicates the primary's
mudler Sep 28, 2026
c524b7b
fix(parakeet-cpp): cap live scene sound score retention
mudler Sep 28, 2026
6eec4f8
fix(parakeet-cpp): merge diarization segments per speaker, harden Dia…
mudler Sep 28, 2026
7eac36f
fix(parakeet-cpp): harden SoundDetection's engine checks
mudler Sep 28, 2026
a63b37e
test(parakeet-cpp): cover a mid-session scene feed failure
mudler Sep 28, 2026
39a9934
docs: fix the parakeet-cpp companion role table and realtime scene docs
mudler Sep 28, 2026
f1f33c7
fix(parakeet-cpp): use CED's real index for Chicken, rooster
mudler Sep 28, 2026
f85fe9a
fix(realtime): call the test event accessor
localai-org-maint-bot Sep 29, 2026
54a869c
chore(parakeet-cpp): pin parakeet.cpp master with sound events
mudler Sep 29, 2026
1dd8b73
chore(parakeet-cpp): pin parakeet.cpp with ced.cpp on main
mudler Sep 29, 2026
a22322a
feat(transcription): carry speaker labels on words and streamed segments
mudler Sep 28, 2026
2cc809e
feat(importers): detect the parakeet.cpp diarization GGUF
mudler Sep 29, 2026
5c3ba9d
fix(config): advertise diarization and sound detection for parakeet-cpp
mudler Sep 29, 2026
5f956b3
feat(parakeet-cpp): label transcript segments with the diarization co…
mudler Sep 29, 2026
1aab6ff
feat(realtime): speaker segments from committed-turn transcription
mudler Sep 29, 2026
f82120b
feat(gallery): add parakeet-cpp-realtime-scene-tdt
mudler Sep 29, 2026
febca11
feat(gallery): add CED-Base variants of the parakeet-cpp scene models
mudler Sep 29, 2026
7b91c7f
fix(config): register pipeline.diarization in the config metadata
mudler Sep 29, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions backend/backend.proto
Original file line number Diff line number Diff line change
Expand Up @@ -641,12 +641,29 @@ message TranscriptLiveResponse {
repeated TranscriptWord words = 4; // words finalized by this feed (stream-relative ns)
TranscriptResult final_result = 5; // terminal message only, after the send side closes
bool eob = 6; // <EOB> fired: a backchannel ("uh-huh") ended — NOT a turn boundary
repeated LiveSpeakerSegment speakers = 7; // closed speaker segments from a companion diarization/scene stream
repeated LiveSoundEvent sounds = 8; // closed sound events from a companion sound/scene stream
}

message LiveSpeakerSegment {
string speaker = 1; // decimal speaker index
int64 start = 2; // stream-relative nanoseconds
int64 end = 3;
}

message LiveSoundEvent {
string label = 1;
int32 index = 2;
float peak = 3;
int64 start = 4; // stream-relative nanoseconds
int64 end = 5;
}

message TranscriptWord {
int64 start = 1;
int64 end = 2;
string text = 3;
string speaker = 4; // backend speaker label when diarizing; empty otherwise
}

message TranscriptSegment {
Expand Down
4 changes: 2 additions & 2 deletions backend/go/parakeet-cpp/Makefile
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
# parakeet-cpp backend Makefile.
#
# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
# Upstream pin lives below as PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea
# (.github/bump_deps.sh) can find and update it - matches the
# whisper.cpp / ds4 / vibevoice-cpp convention.
#
Expand All @@ -15,7 +15,7 @@
# That's what the L0 smoke test uses. The default target below does the
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.

PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp

GOCMD?=go
Expand Down
328 changes: 328 additions & 0 deletions backend/go/parakeet-cpp/diarize.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,328 @@
package main

import (
"encoding/json"
"fmt"
"sort"
"strconv"
"strings"

"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)

// diarizeSegmentJSON mirrors one element of parakeet_capi_diarize_pcm's
// "segments" array: {"speaker":0,"start":0.50,"end":5.52}.
type diarizeSegmentJSON struct {
Speaker int `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
}

// diarizePCMDoc mirrors the document parakeet_capi_diarize_pcm returns.
// "speakers" is the model's CAPACITY (e.g. 8 for Nemotron-3-Diarization),
// not the count of speakers actually present, so it is not read here; the
// response's num_speakers is computed from distinct segment labels instead.
type diarizePCMDoc struct {
Segments []diarizeSegmentJSON `json:"segments"`
}

// diarizeUtteranceJSON mirrors one element of
// parakeet_capi_transcribe_and_diarize_json's "utterances" array. Speaker is
// -1 when no diarized speaker overlaps the utterance.
type diarizeUtteranceJSON struct {
Speaker int `json:"speaker"`
Text string `json:"text"`
Start float64 `json:"start"`
End float64 `json:"end"`
}

// transcribeAndDiarizeDoc mirrors the document
// parakeet_capi_transcribe_and_diarize_json returns. Only "utterances" is
// consumed here; the per-word "words" detail belongs to a speaker-attributed
// transcript RPC, not Diarize.
type transcribeAndDiarizeDoc struct {
Utterances []diarizeUtteranceJSON `json:"utterances"`
}

// speakerLabel renders a 0-based speaker index as the decimal string
// DiarizeSegment.speaker documents, or "unknown" for -1 (no diarized speaker
// overlaps this utterance; only transcribe_and_diarize_json can report this).
func speakerLabel(speaker int) string {
if speaker < 0 {
return "unknown"
}
return strconv.Itoa(speaker)
}

// unsupportedDiarizeFields names the DiarizeRequest fields Sortformer has no
// equivalent for: it is an end-to-end model with a fixed speaker capacity and
// no clustering stage, so there is no config knob to target a speaker count
// or a clustering distance. Logged rather than rejected, so a request naming
// one of these still gets the diarization it can have.
func unsupportedDiarizeFields(req *pb.DiarizeRequest) []string {
var out []string
if req.GetNumSpeakers() != 0 {
out = append(out, "num_speakers")
}
if req.GetMinSpeakers() != 0 {
out = append(out, "min_speakers")
}
if req.GetMaxSpeakers() != 0 {
out = append(out, "max_speakers")
}
if req.GetClusteringThreshold() != 0 {
out = append(out, "clustering_threshold")
}
return out
}

// Diarize labels who spoke when in the audio at req.Dst, using the loaded
// diarization model (p.diarCtx). When req.IncludeText is set and an ASR
// companion (p.ctxPtr) is loaded, each segment also carries its transcript
// (parakeet_capi_transcribe_and_diarize_json, one utterance per speaker
// turn); otherwise, or when no ASR companion is loaded, segments carry no
// text (parakeet_capi_diarize_pcm) and no error is raised.
func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error) {
if p.diarCtx == 0 {
return pb.DiarizeResponse{}, status.Error(codes.FailedPrecondition,
"parakeet-cpp: model is not a diarization model")
}
if CppDiarizePCM == nil {
return pb.DiarizeResponse{}, status.Error(codes.Unimplemented,
"parakeet-cpp: loaded libparakeet.so has no diarization support (parakeet_capi_diarize_pcm missing)")
}
if req.GetDst() == "" {
return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument,
"parakeet-cpp: DiarizeRequest.dst (audio path) is required")
}

if dropped := unsupportedDiarizeFields(req); len(dropped) > 0 {
xlog.Debug("parakeet-cpp: ignoring diarization request fields Sortformer has no equivalent for",
"fields", dropped)
}

pcm, duration, err := decodeWavMono16k(req.GetDst())
if err != nil {
return pb.DiarizeResponse{}, status.Errorf(codes.InvalidArgument, "parakeet-cpp: decode audio: %s", err)
}
if len(pcm) == 0 {
return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument, "parakeet-cpp: empty audio")
}

wantText := req.GetIncludeText() && p.ctxPtr != 0 && CppTranscribeAndDiarizeJSON != nil

raw, err := p.diarizeCall(pcm, wantText)
if err != nil {
return pb.DiarizeResponse{}, err
}
segments, err := parseDiarizeDoc(raw, wantText)
if err != nil {
return pb.DiarizeResponse{}, err
}

segments = applyDurationFilters(segments, req.GetMinDurationOn(), req.GetMinDurationOff())
renumberDiarizeSegments(segments)

return pb.DiarizeResponse{
Segments: segments,
NumSpeakers: distinctDiarizeSpeakers(segments),
Duration: duration,
}, nil
}

// diarizeCall runs the single C call Diarize needs (transcribe_and_diarize_json
// when wantText, else diarize_pcm) under engineMu, and returns the raw JSON
// document. p.diarCtx (and, on the include_text path, p.ctxPtr) is re-checked
// under the lock before the C call: Diarize's own p.diarCtx==0/wantText checks
// run before this lock is taken, so a Free() racing in between (which zeroes
// those fields under the same engineMu) would otherwise reach the C side with
// a freed context. last_error is ctx-shared, so it is read under the same
// lock as the failing call.
func (p *ParakeetCpp) diarizeCall(pcm []float32, wantText bool) (string, error) {
p.engineMu.Lock()
defer p.engineMu.Unlock()

if p.diarCtx == 0 || (wantText && p.ctxPtr == 0) {
return "", grpcerrors.ModelNotLoaded("parakeet-cpp")
}

var cstr uintptr
if wantText {
cstr = CppTranscribeAndDiarizeJSON(p.ctxPtr, p.diarCtx, &pcm[0], int32(len(pcm)), 16000)
} else {
cstr = CppDiarizePCM(p.diarCtx, &pcm[0], int32(len(pcm)), 16000)
}
if cstr == 0 {
return "", fmt.Errorf("parakeet-cpp: diarize failed: %s", diarizeLastError(p, wantText))
}
raw := goStringFromCPtr(cstr)
CppFreeString(cstr)
return raw, nil
}

// diarizeLastError reads last_error off p.diarCtx and, on the include_text
// path, p.ctxPtr too — the failing call is CppTranscribeAndDiarizeJSON there,
// and either side of the pairing may be the one that set it — then joins
// whichever came back non-empty. Called under the same engineMu as the
// failing call (last_error is ctx-shared state).
func diarizeLastError(p *ParakeetCpp, wantText bool) string {
var msgs []string
if m := CppLastError(p.diarCtx); m != "" {
msgs = append(msgs, m)
}
if wantText {
if m := CppLastError(p.ctxPtr); m != "" {
msgs = append(msgs, m)
}
}
if len(msgs) == 0 {
return "unknown error"
}
return strings.Join(msgs, "; ")
}

// parseDiarizeDoc decodes the raw JSON diarizeCall returned into
// DiarizeSegments (without ids: renumberDiarizeSegments assigns those after
// filtering).
func parseDiarizeDoc(raw string, wantText bool) ([]*pb.DiarizeSegment, error) {
if wantText {
var doc transcribeAndDiarizeDoc
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err)
}
segs := make([]*pb.DiarizeSegment, 0, len(doc.Utterances))
for _, u := range doc.Utterances {
segs = append(segs, &pb.DiarizeSegment{
Start: float32(u.Start),
End: float32(u.End),
Speaker: speakerLabel(u.Speaker),
Text: u.Text,
})
}
return segs, nil
}

var doc diarizePCMDoc
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err)
}
segs := make([]*pb.DiarizeSegment, 0, len(doc.Segments))
for _, s := range doc.Segments {
segs = append(segs, &pb.DiarizeSegment{
Start: float32(s.Start),
End: float32(s.End),
Speaker: speakerLabel(s.Speaker),
})
}
return segs, nil
}

// applyDurationFilters applies the request's postprocessing knobs, in the
// order NeMo's diarization postprocessing does: merge first
// (min_duration_off), then drop short segments (min_duration_on) — dropping
// first would leave short gaps unmerged that the drop step just created.
// Segments are assumed sorted by start time, as parakeet_capi_diarize_pcm and
// parakeet_capi_transcribe_and_diarize_json document. A non-positive value
// disables that filter (the proto's "0 = backend default" reads here as "no
// filtering").
func applyDurationFilters(segs []*pb.DiarizeSegment, minOn, minOff float32) []*pb.DiarizeSegment {
segs = mergeCloseSegments(segs, minOff)
segs = dropShortSegments(segs, minOn)
return segs
}

// mergeCloseSegments merges SAME-SPEAKER segments separated by a gap shorter
// than minOff into one segment spanning both (and concatenating any text).
// Segments from different speakers are never merged, regardless of gap: the
// gap only ever means "the same speaker paused", never "two speakers are
// actually one".
//
// Merging runs per speaker rather than on the single start-sorted list: two
// segments of the same speaker are not necessarily adjacent in that list once
// another speaker's turn falls between them (A, B, A), and a start-sorted
// walk would then never compare the two A's at all. Grouping by speaker first
// keeps each group's own start order (segs is assumed start-sorted, as
// parakeet_capi_diarize_pcm and parakeet_capi_transcribe_and_diarize_json
// document), merges within the group, then the merged segments are re-sorted
// by start so interleaved speakers come back out in timeline order.
func mergeCloseSegments(segs []*pb.DiarizeSegment, minOff float32) []*pb.DiarizeSegment {
if minOff <= 0 || len(segs) < 2 {
return segs
}

bySpeaker := make(map[string][]*pb.DiarizeSegment)
var order []string // first-seen speaker order, for a deterministic group walk
for _, s := range segs {
if _, ok := bySpeaker[s.GetSpeaker()]; !ok {
order = append(order, s.GetSpeaker())
}
bySpeaker[s.GetSpeaker()] = append(bySpeaker[s.GetSpeaker()], s)
}

out := make([]*pb.DiarizeSegment, 0, len(segs))
for _, speaker := range order {
group := bySpeaker[speaker]
merged := make([]*pb.DiarizeSegment, 0, len(group))
merged = append(merged, group[0])
for _, s := range group[1:] {
prev := merged[len(merged)-1]
if s.GetStart()-prev.GetEnd() < minOff {
if s.GetEnd() > prev.GetEnd() {
prev.End = s.End
}
if s.GetText() != "" {
if prev.GetText() != "" {
prev.Text = prev.GetText() + " " + s.GetText()
} else {
prev.Text = s.GetText()
}
}
continue
}
merged = append(merged, s)
}
out = append(out, merged...)
}

sort.Slice(out, func(i, j int) bool { return out[i].GetStart() < out[j].GetStart() })
return out
}

// dropShortSegments discards segments shorter than minOn.
func dropShortSegments(segs []*pb.DiarizeSegment, minOn float32) []*pb.DiarizeSegment {
if minOn <= 0 {
return segs
}
out := make([]*pb.DiarizeSegment, 0, len(segs))
for _, s := range segs {
if s.GetEnd()-s.GetStart() < minOn {
continue
}
out = append(out, s)
}
return out
}

// renumberDiarizeSegments assigns sequential ids (0..) to the final segment
// list, after filtering may have dropped or merged entries.
func renumberDiarizeSegments(segs []*pb.DiarizeSegment) {
for i, s := range segs {
s.Id = int32(i)
}
}

// distinctDiarizeSpeakers counts the distinct speaker labels present in segs.
// This is what DiarizeResponse.num_speakers documents — the count of speakers
// actually present in the result — and is NOT the diarize_pcm JSON's
// top-level "speakers" field, which reports the model's fixed capacity.
func distinctDiarizeSpeakers(segs []*pb.DiarizeSegment) int32 {
seen := make(map[string]struct{}, len(segs))
for _, s := range segs {
seen[s.GetSpeaker()] = struct{}{}
}
return int32(len(seen))
}
Loading
Loading