~/wiki

Tensor Processing

Confiance : high
tensor-processingonnx-modelsdata-transformationtensor-shapesmel-spectrogramaudio-processingmachine-learning-inferencebuffer-managementrolling-buffersmulti-dimensional-arraysshape-manipulationswift-mlonnx-runtime

Mathematical operations on multi-dimensional arrays (tensors) that form the foundation of machine learning inference pipelines. Critical for real-time audio processing where tensor shape transformations and buffer management directly impact performance and accuracy.

Core Concepts

Tensor Dimensions and Shapes

// Common tensor shapes in audio ML pipelines:
let audioSamples: [Float] = [...]        // Shape: [N] - 1D array
let melSpectrogram: Float = [...]    // Shape: [T, F] - Time × Frequency
let batchedInput: [Float] = [...]    // Shape: [B, T, F] - Batch × Time × Frequency

Shape Transformations

Critical operations for preparing data between model stages:

// Squeeze: Remove dimensions of size 1
// [T, 1, 1, 32] → [T, 32]
func squeeze4Dto2D(_ tensor: [[Float]]) -> Float {
    return tensor.map { timeFrame in
        return timeFrame[0][0]  // Extract [32] from [1, 1, 32]
    }
}

// Expand: Add batch dimension
// [16, 96] → [1, 16, 96]  
func addBatchDimension(_ tensor: Float) -> [Float] {
    return [tensor]
}

ONNX Runtime Integration

Input/Output Tensor Handling

// Create input tensor from Swift array
let inputData = Data(bytes: floatArray, count: floatArray.count * 4)
let inputTensor = try ORTValue(
    tensorData: NSMutableData(data: inputData),
    elementType: .float,
    shape: [1, timeFrames, melBins]
)

// Extract output tensor data
let outputTensor = results[outputName]
let outputData = try outputTensor.tensorData() as Data
let outputFloats = outputData.withUnsafeBytes {
    Array($0.bindMemory(to: Float.self))
}

Dynamic Shape Handling

class OpenWakeWordPipeline {
    func processAudioChunk(_ samples: [Float]) -> Float? {
        // Stage 1: Audio → Mel-Spectrogram [N] → [T, 32]
        let melFrames = try melSpectrogramModel.run(samples)
        let squeezedMel = squeeze4Dto2D(melFrames)  // [T, 1, 1, 32] → [T, 32]
        let normalizedMel = squeezedMel.map { $0.map { $0 / 10 + 2 } }
        
        // Stage 2: Mel → Embedding [76, 32] → [96]
        melFrameBuffer.append(contentsOf: normalizedMel)
        guard melFrameBuffer.count >= 76 else { return nil }
        
        let melWindow = Array(melFrameBuffer.suffix(76))
        let batchedMel = [melWindow]  // Add batch dimension: [1, 76, 32]
        let embedding = try embeddingModel.run(batchedMel)
        let flatEmbedding = squeeze3Dto1D(embedding)  // [1, 1, 96] → [96]
        
        // Stage 3: Embeddings → Classification [16, 96] → [1]
        embeddingBuffer.append(flatEmbedding)
        guard embeddingBuffer.count >= 16 else { return nil }
        
        let embeddingSequence = Array(embeddingBuffer.suffix(16))
        let batchedSequence = [embeddingSequence]  // [1, 16, 96]
        let confidence = try classifierModel.run(batchedSequence)
        
        return confidence[0]  // Extract scalar confidence score
    }
}

Real-Time Buffer Management

Rolling Tensor Windows

// Maintain rolling windows at different tensor dimensions
class MultiDimensionalRollingBuffer {
    private var audioBuffer: [Float] = []              // 1D: [N]
    private var melFrameBuffer: Float = []         // 2D: [T, F]  
    private var embeddingBuffer: Float = []        // 2D: [T, E]
    
    func appendAudio(_ samples: [Float]) {
        audioBuffer.append(contentsOf: samples)
        if audioBuffer.count > maxAudioSamples {
            audioBuffer.removeFirst(audioBuffer.count - maxAudioSamples)
        }
    }
    
    func appendMelFrames(_ frames: Float) {
        melFrameBuffer.append(contentsOf: frames)
        if melFrameBuffer.count > maxMelFrames {
            melFrameBuffer.removeFirst(melFrameBuffer.count - maxMelFrames)
        }
    }
    
    func getAudioWindow(_ samples: Int) -> [Float] {
        return Array(audioBuffer.suffix(samples))
    }
    
    func getMelWindow(_ frames: Int) -> Float {
        return Array(melFrameBuffer.suffix(frames))
    }
}

Memory-Efficient Tensor Operations

// Avoid unnecessary copying for large tensors
extension Array where Element == Float {
    func withContiguousStorage<R>(_ body: (UnsafeBufferPointer<Float>) -> R) -> R {
        return self.withUnsafeBufferPointer(body)
    }
}

// Process tensors in-place when possible
func normalizeInPlace(_ tensor: inout Float) {
    for i in 0..<tensor.count {
        for j in 0..<tensor[i].count {
            tensor[i][j] = tensor[i][j] / 10.0 +