Files
FluidAudio/Sources/FluidAudioCLI/Utils/RTTMParser.swift
7fd5ac5446 pyannote community-1 model for offline speaker diarization pipeline (#150)
### Why is this change needed?
<!-- Explain the motivation for this change. What problem does it solve?
-->

Keeping the streaming one around as the VBx and AHC clustering gets
pretty expensive after 30mins of audio and running it constantly gets
expensive. Its still possible to support clustering between files but
will save that for another PR.

Pyannote's Bench mark is around 11% - i increased steps to 0.2s instead
of 0.1 to double the speed but also selective fp16 results in more
operations to run on ANE but also means that we lose some precision.

```
Average DER: 14.95% | Median DER: 10.89% | Average JER: 39.27% | Median JER: 40.74% (collar=0.25s, ignoreOverlap=True)
Average RTFx: 139.63 (from 232 clips)
Metrics summary saved to: /Users/brandonweng/FluidAudioDatasets/voxconverse/metrics/test_metrics_release.json
Completed. New results: 232, Skipped existing: 0, Total attempted: 232
```

See benchmark.md for more info but compared to Pytorch model, we are
100x faster than the CPU version and ~6x faster compared to the mps
backend on mb pro 4

---------

Co-authored-by: claude[bot] <209825114+claude[bot]@users.noreply.github.com>
Co-authored-by: Brandon Weng <BrandonWeng@users.noreply.github.com>
Co-authored-by: Alex <36247722+Alex-Wengg@users.noreply.github.com>
Co-authored-by: Alex-Wengg <hanweng9@gmail.com>
2025-10-22 15:11:57 -04:00

66 lines
2.0 KiB
Swift

#if os(macOS)
import FluidAudio
import Foundation
enum RTTMParserError: Error, LocalizedError {
case fileNotFound(String)
case invalidLine(String)
var errorDescription: String? {
switch self {
case .fileNotFound(let path):
return "RTTM file not found at \(path)"
case .invalidLine(let line):
return "Invalid RTTM line: \(line)"
}
}
}
/// Lightweight RTTM parser for converting ground-truth annotations into `TimedSpeakerSegment`s.
enum RTTMParser {
static func loadSegments(from path: String) throws -> [TimedSpeakerSegment] {
guard FileManager.default.fileExists(atPath: path) else {
throw RTTMParserError.fileNotFound(path)
}
let contents = try String(contentsOfFile: path, encoding: .utf8)
var segments: [TimedSpeakerSegment] = []
for rawLine in contents.components(separatedBy: .newlines) {
let line = rawLine.trimmingCharacters(in: .whitespaces)
if line.isEmpty || line.hasPrefix("#") {
continue
}
let fields = line.split(whereSeparator: { $0.isWhitespace })
guard fields.count >= 8, fields[0] == "SPEAKER" else {
throw RTTMParserError.invalidLine(line)
}
guard
let start = Float(fields[3]),
let duration = Float(fields[4])
else {
throw RTTMParserError.invalidLine(line)
}
let speakerId = String(fields[7])
let endTime = start + duration
segments.append(
TimedSpeakerSegment(
speakerId: speakerId,
embedding: [],
startTimeSeconds: start,
endTimeSeconds: endTime,
qualityScore: 1.0
)
)
}
return segments.sorted { $0.startTimeSeconds < $1.startTimeSeconds }
}
}
#endif