File size: 1,369 Bytes
5189a69 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 | {
"model_type": "pyannote-segmentation",
"architecture": "sincnet-bilstm-classifier",
"framework": "mlx",
"original_model": "pyannote/segmentation-3.0",
"conversion_date": "2026-01-16",
"parameters": 1473515,
"model_size_mb": 5.6,
"input": {
"type": "audio",
"sample_rate": 16000,
"channels": 1,
"format": "waveform",
"dtype": "float32"
},
"output": {
"type": "logits",
"num_classes": 7,
"frame_duration_ms": 17,
"activation": "log_softmax"
},
"architecture_details": {
"sincnet": {
"num_filters": 80,
"kernel_size": 251,
"num_layers": 3
},
"lstm": {
"num_layers": 4,
"hidden_size": 128,
"bidirectional": true,
"output_size": 256
},
"classifier": {
"hidden_dim": 128,
"num_classes": 7
}
},
"validation": {
"pytorch_correlation": 0.886,
"sincnet_correlation": 0.9999999999,
"lstm_correlation": 0.999,
"component_validation": "perfect",
"status": "production_ready"
},
"performance": {
"platform": "apple_silicon",
"backend": "metal",
"memory_model": "unified",
"gpu_accelerated": true
},
"license": "MIT",
"tags": [
"speaker-diarization",
"audio",
"mlx",
"apple-silicon",
"pyannote",
"sincnet",
"lstm",
"speaker-segmentation"
]
}
|