mirror of
https://github.com/mudler/LocalAI.git
synced 2026-08-01 19:09:42 -04:00
Three measured HTTP-layer regressions on a live deployment, fixed together
because they all shape the bytes on the wire.
1. No compression. The server sent no Content-Encoding regardless of what
the client asked for, confirmed with curl straight at 127.0.0.1:8080 so
it was not an ingress artefact. Adds gzip middleware, on by default and
configurable via LOCALAI_DISABLE_HTTP_COMPRESSION and
LOCALAI_HTTP_COMPRESSION_MIN_LENGTH (default 1024 bytes so tiny bodies
are not wastefully wrapped). Streaming routes are skipped explicitly:
an SSE Accept header, a WebSocket upgrade, and the completion / SSE /
log-tail path prefixes, because whether a completion request streams is
decided by the request body, which the middleware runs too early to see.
Already-compressed formats (woff2, png, mp4, ...) are skipped too; gzip
made those marginally larger. Measured over the embedded React build:
JS+CSS 2815 KB raw to 808 KB gzipped (3.48x).
2. No cache headers on content-hashed assets. Vite hashes the filenames,
so a given /assets/ URL can never change content, yet they shipped with
no Cache-Control, ETag or Last-Modified, and the browser re-fetched the
whole bundle on every navigation with no conditional request available.
/assets/* now carries public, max-age=31536000, immutable. index.html
stays no-cache so a deploy is picked up, and the unhashed locale JSONs
get a short TTL rather than the immutable one.
3. Unbounded trace endpoints. /api/traces returned 21,033,606 bytes in
4.65s and /api/backend-traces 3,471,682 bytes in 1.50s, and the admin
UI polls both every few seconds. The ring buffer holds up to 1024
entries, each embedding full input_text payloads. Both list endpoints
now take limit / offset / full, default to 50 entries, and strip the
heavy fields (request and response bodies plus headers for API traces,
body and data for backend traces) unless full=true. Every trace gets a
process-lifetime ID and GET /api/traces/{id} and
/api/backend-traces/{id} serve the full record, which is what the UI
fetches when a row is expanded. The list body stays a JSON array;
paging metadata rides in X-Total-Count, X-Trace-Offset and
X-Trace-Limit. Reproducing the live shape in a test, the polled payload
goes from 21,131,097 bytes to 7,201 bytes.
Assisted-by: Claude:claude-opus-4-8 [Claude Code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
88 lines
3.3 KiB
JavaScript
88 lines
3.3 KiB
JavaScript
import { test, expect } from './coverage-fixtures.js'
|
|
|
|
// Audio snippets on the Traces page must play through a blob: object URL —
|
|
// the CSP's connect-src allows blob: but not data:, and the waveform peaks
|
|
// renderer fetch()es the player src — and must degrade to a readable note
|
|
// (not a broken player) when the stored payload is the "<truncated: N bytes>"
|
|
// marker an older server stamped into oversized fields.
|
|
|
|
// Minimal valid 16 kHz mono 16-bit PCM WAV (0.1s 440 Hz sine), base64-encoded.
|
|
function wavBase64(samples = 1600, rate = 16000) {
|
|
const dataSize = samples * 2
|
|
const buf = Buffer.alloc(44 + dataSize)
|
|
buf.write('RIFF', 0)
|
|
buf.writeUInt32LE(36 + dataSize, 4)
|
|
buf.write('WAVE', 8)
|
|
buf.write('fmt ', 12)
|
|
buf.writeUInt32LE(16, 16)
|
|
buf.writeUInt16LE(1, 20) // PCM
|
|
buf.writeUInt16LE(1, 22) // mono
|
|
buf.writeUInt32LE(rate, 24)
|
|
buf.writeUInt32LE(rate * 2, 28)
|
|
buf.writeUInt16LE(2, 32)
|
|
buf.writeUInt16LE(16, 34)
|
|
buf.write('data', 36)
|
|
buf.writeUInt32LE(dataSize, 40)
|
|
for (let i = 0; i < samples; i++) {
|
|
buf.writeInt16LE(Math.round(8000 * Math.sin((2 * Math.PI * 440 * i) / rate)), 44 + i * 2)
|
|
}
|
|
return buf.toString('base64')
|
|
}
|
|
|
|
function transcriptionTrace(audioWavBase64) {
|
|
return {
|
|
type: 'transcription',
|
|
timestamp: Date.now() * 1_000_000,
|
|
model_name: 'parakeet-test',
|
|
summary: 'transcribed utterance',
|
|
duration: 500_000_000,
|
|
error: null,
|
|
data: {
|
|
audio_wav_base64: audioWavBase64,
|
|
audio_duration_s: 0.1,
|
|
audio_snippet_s: 0.1,
|
|
audio_sample_rate: 16000,
|
|
audio_samples: 1600,
|
|
audio_rms_dbfs: -12.0,
|
|
audio_peak_dbfs: -6.0,
|
|
audio_dc_offset: 0,
|
|
},
|
|
}
|
|
}
|
|
|
|
async function openBackendTraceRow(page, traces) {
|
|
await page.route('**/api/traces?*', (route) => {
|
|
route.fulfill({ contentType: 'application/json', body: JSON.stringify([]) })
|
|
})
|
|
await page.route('**/api/backend-traces?*', (route) => {
|
|
route.fulfill({ contentType: 'application/json', body: JSON.stringify(traces) })
|
|
})
|
|
await page.goto('/app/traces')
|
|
await expect(page.locator('text=Tracing is')).toBeVisible({ timeout: 10_000 })
|
|
await page.locator('button', { hasText: 'Backend Traces' }).click()
|
|
await page.locator('td', { hasText: 'parakeet-test' }).first().click()
|
|
}
|
|
|
|
test.describe('Traces - Audio Snippets', () => {
|
|
test('plays a clip through a blob: URL, not a CSP-blocked data: URL', async ({ page }) => {
|
|
await openBackendTraceRow(page, [transcriptionTrace(wavBase64())])
|
|
|
|
// The expanded row carries the snippet metrics and a player whose source
|
|
// is an object URL (connect-src allows blob:, so the peaks fetch works).
|
|
await expect(page.locator('text=Audio Snippet')).toBeVisible()
|
|
const audio = page.locator('audio')
|
|
await expect(audio).toHaveCount(1)
|
|
const src = await audio.getAttribute('src')
|
|
expect(src).toMatch(/^blob:/)
|
|
await expect(page.getByTestId('audio-snippet-unavailable')).toHaveCount(0)
|
|
})
|
|
|
|
test('shows a readable note instead of a broken player for truncated payloads', async ({ page }) => {
|
|
await openBackendTraceRow(page, [transcriptionTrace('<truncated: 281660 bytes>')])
|
|
|
|
await expect(page.locator('text=Audio Snippet')).toBeVisible()
|
|
await expect(page.getByTestId('audio-snippet-unavailable')).toBeVisible()
|
|
await expect(page.locator('audio')).toHaveCount(0)
|
|
})
|
|
})
|