Tensor Processing
Confiance : high
tensor-processingonnx-modelsdata-transformationtensor-shapesmel-spectrogramaudio-processingmachine-learning-inferencebuffer-managementrolling-buffersmulti-dimensional-arraysshape-manipulationswift-mlonnx-runtime
Mathematical operations on multi-dimensional arrays (tensors) that form the foundation of machine learning inference pipelines. Critical for real-time audio processing where tensor shape transformations and buffer management directly impact performance and accuracy.
Core Concepts
Tensor Dimensions and Shapes
// Common tensor shapes in audio ML pipelines:
let audioSamples: [Float] = [...] // Shape: [N] - 1D array
let melSpectrogram: Float = [...] // Shape: [T, F] - Time × Frequency
let batchedInput: [Float] = [...] // Shape: [B, T, F] - Batch × Time × Frequency
Shape Transformations
Critical operations for preparing data between model stages:
// Squeeze: Remove dimensions of size 1
// [T, 1, 1, 32] → [T, 32]
func squeeze4Dto2D(_ tensor: [[Float]]) -> Float {
return tensor.map { timeFrame in
return timeFrame[0][0] // Extract [32] from [1, 1, 32]
}
}
// Expand: Add batch dimension
// [16, 96] → [1, 16, 96]
func addBatchDimension(_ tensor: Float) -> [Float] {
return [tensor]
}
ONNX Runtime Integration
Input/Output Tensor Handling
// Create input tensor from Swift array
let inputData = Data(bytes: floatArray, count: floatArray.count * 4)
let inputTensor = try ORTValue(
tensorData: NSMutableData(data: inputData),
elementType: .float,
shape: [1, timeFrames, melBins]
)
// Extract output tensor data
let outputTensor = results[outputName]
let outputData = try outputTensor.tensorData() as Data
let outputFloats = outputData.withUnsafeBytes {
Array($0.bindMemory(to: Float.self))
}
Dynamic Shape Handling
class OpenWakeWordPipeline {
func processAudioChunk(_ samples: [Float]) -> Float? {
// Stage 1: Audio → Mel-Spectrogram [N] → [T, 32]
let melFrames = try melSpectrogramModel.run(samples)
let squeezedMel = squeeze4Dto2D(melFrames) // [T, 1, 1, 32] → [T, 32]
let normalizedMel = squeezedMel.map { $0.map { $0 / 10 + 2 } }
// Stage 2: Mel → Embedding [76, 32] → [96]
melFrameBuffer.append(contentsOf: normalizedMel)
guard melFrameBuffer.count >= 76 else { return nil }
let melWindow = Array(melFrameBuffer.suffix(76))
let batchedMel = [melWindow] // Add batch dimension: [1, 76, 32]
let embedding = try embeddingModel.run(batchedMel)
let flatEmbedding = squeeze3Dto1D(embedding) // [1, 1, 96] → [96]
// Stage 3: Embeddings → Classification [16, 96] → [1]
embeddingBuffer.append(flatEmbedding)
guard embeddingBuffer.count >= 16 else { return nil }
let embeddingSequence = Array(embeddingBuffer.suffix(16))
let batchedSequence = [embeddingSequence] // [1, 16, 96]
let confidence = try classifierModel.run(batchedSequence)
return confidence[0] // Extract scalar confidence score
}
}
Real-Time Buffer Management
Rolling Tensor Windows
// Maintain rolling windows at different tensor dimensions
class MultiDimensionalRollingBuffer {
private var audioBuffer: [Float] = [] // 1D: [N]
private var melFrameBuffer: Float = [] // 2D: [T, F]
private var embeddingBuffer: Float = [] // 2D: [T, E]
func appendAudio(_ samples: [Float]) {
audioBuffer.append(contentsOf: samples)
if audioBuffer.count > maxAudioSamples {
audioBuffer.removeFirst(audioBuffer.count - maxAudioSamples)
}
}
func appendMelFrames(_ frames: Float) {
melFrameBuffer.append(contentsOf: frames)
if melFrameBuffer.count > maxMelFrames {
melFrameBuffer.removeFirst(melFrameBuffer.count - maxMelFrames)
}
}
func getAudioWindow(_ samples: Int) -> [Float] {
return Array(audioBuffer.suffix(samples))
}
func getMelWindow(_ frames: Int) -> Float {
return Array(melFrameBuffer.suffix(frames))
}
}
Memory-Efficient Tensor Operations
// Avoid unnecessary copying for large tensors
extension Array where Element == Float {
func withContiguousStorage<R>(_ body: (UnsafeBufferPointer<Float>) -> R) -> R {
return self.withUnsafeBufferPointer(body)
}
}
// Process tensors in-place when possible
func normalizeInPlace(_ tensor: inout Float) {
for i in 0..<tensor.count {
for j in 0..<tensor[i].count {
tensor[i][j] = tensor[i][j] / 10.0 +