Files
LocalAI/core/http/react-ui/e2e/traces-audio.spec.js
mudler's LocalAI [bot] ff299df453 perf(http): gzip responses, cache hashed assets, bound the trace endpoints (#11056)
Three measured HTTP-layer regressions on a live deployment, fixed together
because they all shape the bytes on the wire.

1. No compression. The server sent no Content-Encoding regardless of what
   the client asked for, confirmed with curl straight at 127.0.0.1:8080 so
   it was not an ingress artefact. Adds gzip middleware, on by default and
   configurable via LOCALAI_DISABLE_HTTP_COMPRESSION and
   LOCALAI_HTTP_COMPRESSION_MIN_LENGTH (default 1024 bytes so tiny bodies
   are not wastefully wrapped). Streaming routes are skipped explicitly:
   an SSE Accept header, a WebSocket upgrade, and the completion / SSE /
   log-tail path prefixes, because whether a completion request streams is
   decided by the request body, which the middleware runs too early to see.
   Already-compressed formats (woff2, png, mp4, ...) are skipped too; gzip
   made those marginally larger. Measured over the embedded React build:
   JS+CSS 2815 KB raw to 808 KB gzipped (3.48x).

2. No cache headers on content-hashed assets. Vite hashes the filenames,
   so a given /assets/ URL can never change content, yet they shipped with
   no Cache-Control, ETag or Last-Modified, and the browser re-fetched the
   whole bundle on every navigation with no conditional request available.
   /assets/* now carries public, max-age=31536000, immutable. index.html
   stays no-cache so a deploy is picked up, and the unhashed locale JSONs
   get a short TTL rather than the immutable one.

3. Unbounded trace endpoints. /api/traces returned 21,033,606 bytes in
   4.65s and /api/backend-traces 3,471,682 bytes in 1.50s, and the admin
   UI polls both every few seconds. The ring buffer holds up to 1024
   entries, each embedding full input_text payloads. Both list endpoints
   now take limit / offset / full, default to 50 entries, and strip the
   heavy fields (request and response bodies plus headers for API traces,
   body and data for backend traces) unless full=true. Every trace gets a
   process-lifetime ID and GET /api/traces/{id} and
   /api/backend-traces/{id} serve the full record, which is what the UI
   fetches when a row is expanded. The list body stays a JSON array;
   paging metadata rides in X-Total-Count, X-Trace-Offset and
   X-Trace-Limit. Reproducing the live shape in a test, the polled payload
   goes from 21,131,097 bytes to 7,201 bytes.


Assisted-by: Claude:claude-opus-4-8 [Claude Code]

Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
2026-07-22 22:51:25 +02:00

88 lines
3.3 KiB
JavaScript

import { test, expect } from './coverage-fixtures.js'
// Audio snippets on the Traces page must play through a blob: object URL —
// the CSP's connect-src allows blob: but not data:, and the waveform peaks
// renderer fetch()es the player src — and must degrade to a readable note
// (not a broken player) when the stored payload is the "<truncated: N bytes>"
// marker an older server stamped into oversized fields.
// Minimal valid 16 kHz mono 16-bit PCM WAV (0.1s 440 Hz sine), base64-encoded.
function wavBase64(samples = 1600, rate = 16000) {
const dataSize = samples * 2
const buf = Buffer.alloc(44 + dataSize)
buf.write('RIFF', 0)
buf.writeUInt32LE(36 + dataSize, 4)
buf.write('WAVE', 8)
buf.write('fmt ', 12)
buf.writeUInt32LE(16, 16)
buf.writeUInt16LE(1, 20) // PCM
buf.writeUInt16LE(1, 22) // mono
buf.writeUInt32LE(rate, 24)
buf.writeUInt32LE(rate * 2, 28)
buf.writeUInt16LE(2, 32)
buf.writeUInt16LE(16, 34)
buf.write('data', 36)
buf.writeUInt32LE(dataSize, 40)
for (let i = 0; i < samples; i++) {
buf.writeInt16LE(Math.round(8000 * Math.sin((2 * Math.PI * 440 * i) / rate)), 44 + i * 2)
}
return buf.toString('base64')
}
function transcriptionTrace(audioWavBase64) {
return {
type: 'transcription',
timestamp: Date.now() * 1_000_000,
model_name: 'parakeet-test',
summary: 'transcribed utterance',
duration: 500_000_000,
error: null,
data: {
audio_wav_base64: audioWavBase64,
audio_duration_s: 0.1,
audio_snippet_s: 0.1,
audio_sample_rate: 16000,
audio_samples: 1600,
audio_rms_dbfs: -12.0,
audio_peak_dbfs: -6.0,
audio_dc_offset: 0,
},
}
}
async function openBackendTraceRow(page, traces) {
await page.route('**/api/traces?*', (route) => {
route.fulfill({ contentType: 'application/json', body: JSON.stringify([]) })
})
await page.route('**/api/backend-traces?*', (route) => {
route.fulfill({ contentType: 'application/json', body: JSON.stringify(traces) })
})
await page.goto('/app/traces')
await expect(page.locator('text=Tracing is')).toBeVisible({ timeout: 10_000 })
await page.locator('button', { hasText: 'Backend Traces' }).click()
await page.locator('td', { hasText: 'parakeet-test' }).first().click()
}
test.describe('Traces - Audio Snippets', () => {
test('plays a clip through a blob: URL, not a CSP-blocked data: URL', async ({ page }) => {
await openBackendTraceRow(page, [transcriptionTrace(wavBase64())])
// The expanded row carries the snippet metrics and a player whose source
// is an object URL (connect-src allows blob:, so the peaks fetch works).
await expect(page.locator('text=Audio Snippet')).toBeVisible()
const audio = page.locator('audio')
await expect(audio).toHaveCount(1)
const src = await audio.getAttribute('src')
expect(src).toMatch(/^blob:/)
await expect(page.getByTestId('audio-snippet-unavailable')).toHaveCount(0)
})
test('shows a readable note instead of a broken player for truncated payloads', async ({ page }) => {
await openBackendTraceRow(page, [transcriptionTrace('<truncated: 281660 bytes>')])
await expect(page.locator('text=Audio Snippet')).toBeVisible()
await expect(page.getByTestId('audio-snippet-unavailable')).toBeVisible()
await expect(page.locator('audio')).toHaveCount(0)
})
})