Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion backend/cpp/ik-llama-cpp/Makefile
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@

IK_LLAMA_VERSION?=0be97a7a5ad113f33e08729261649ccea2cdc5ff
IK_LLAMA_VERSION?=cb9147fd0d9c08a9a84eee5ac405a73f4e10e3e1
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp

CMAKE_ARGS?=
Expand Down
30 changes: 30 additions & 0 deletions core/http/endpoints/localai/traces.go
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ package localai
import (
"net/http"
"strconv"
"time"

"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/http/middleware"
Expand Down Expand Up @@ -85,6 +86,35 @@ func GetAPITracesEndpoint() echo.HandlerFunc {
}
}

// GetAPITracesSummaryEndpoint returns counted totals over a recent window
// @Summary Summarize recent API traces
// @Description Returns request, failure and latency totals over a recent window, plus a bucketed series for sparklines. Exists so callers wanting three numbers do not have to fetch the whole trace list and count it themselves.
// @Tags monitoring
// @Produce json
// @Param hours query int false "Window in hours (default 24, max 168)"
// @Success 200 {object} middleware.TraceSummary "Counted trace totals"
// @Router /api/traces/summary [get]
func GetAPITracesSummaryEndpoint() echo.HandlerFunc {
return func(c echo.Context) error {
hours := 24
if raw := c.QueryParam("hours"); raw != "" {
if v, err := strconv.Atoi(raw); err == nil && v > 0 {
hours = v
}
}
// A week is plenty for a dashboard, and the trace buffer is bounded
// anyway; an unbounded window would just scan the whole buffer.
if hours > 168 {
hours = 168
}
return c.JSON(http.StatusOK, middleware.GetTracesSummary(time.Duration(hours)*time.Hour, traceSummaryBuckets))
}
}

// Enough columns for a sparkline to show a shape, few enough that each one
// still holds a meaningful count on a quiet installation.
const traceSummaryBuckets = 12

// GetAPITraceEndpoint returns a single API trace with its full payload
// @Summary Get one API trace
// @Description Returns a single captured API exchange, including the request and response bodies omitted from the list response
Expand Down
109 changes: 109 additions & 0 deletions core/http/middleware/trace_summary.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,109 @@
// SPDX-License-Identifier: MIT

package middleware

import (
"math"
"slices"
"time"
)

// TraceSummary is the counted view of the trace buffer.
//
// It exists so a caller that wants "how many, how many failed, how slow" does
// not have to fetch every exchange and count them in the browser. The Operate
// overview needs exactly those three numbers, and the trace list is capped in
// the thousands, so shipping it across the wire to produce a single integer is
// waste that grows with the buffer.
type TraceSummary struct {
Total int `json:"total"`
Errors int `json:"errors"`
P95Millis int64 `json:"p95_ms"`
WindowHours int `json:"window_hours"`
Buckets []TraceBucket `json:"buckets"`
}

// TraceBucket is one column of a sparkline: oldest first, so the series reads
// left to right the way a chart is drawn.
type TraceBucket struct {
Start time.Time `json:"start"`
Count int `json:"count"`
Errors int `json:"errors"`
}

// GetTracesSummary counts the buffered exchanges over the given window.
func GetTracesSummary(window time.Duration, buckets int) TraceSummary {
return summarize(GetTraces(), window, buckets)
}

func summarize(traces []APIExchange, window time.Duration, buckets int) TraceSummary {
if buckets < 1 {
buckets = 1
}
now := time.Now()
cutoff := now.Add(-window)

summary := TraceSummary{
WindowHours: int(window.Hours()),
// Never nil: a nil slice serialises as null and breaks .map() on the
// other side, which is a silent runtime error rather than an empty chart.
Buckets: make([]TraceBucket, buckets),
}

bucketWidth := window / time.Duration(buckets)
for i := range summary.Buckets {
summary.Buckets[i].Start = cutoff.Add(time.Duration(i) * bucketWidth)
}

durations := make([]time.Duration, 0, len(traces))
for _, t := range traces {
if t.Timestamp.Before(cutoff) {
continue
}
summary.Total++
failed := isFailure(t)
if failed {
summary.Errors++
}
durations = append(durations, t.Duration)

// Clamp rather than skip: a request timestamped a hair in the future
// (clock skew, or arriving mid-call) still belongs in the newest column.
idx := int(t.Timestamp.Sub(cutoff) / bucketWidth)
if idx >= buckets {
idx = buckets - 1
}
if idx < 0 {
idx = 0
}
summary.Buckets[idx].Count++
if failed {
summary.Buckets[idx].Errors++
}
}

summary.P95Millis = percentileMillis(durations, 0.95)
return summary
}

// A 4xx is the caller getting it wrong, which is not the installation being
// unhealthy. Only 5xx and a transport-level error count against the runtime.
func isFailure(t APIExchange) bool {
return t.Error != "" || t.Response.Status >= 500
}

func percentileMillis(durations []time.Duration, p float64) int64 {
if len(durations) == 0 {
return 0
}
slices.Sort(durations)
// Nearest-rank: the smallest value at or above the pth percentile.
rank := int(math.Ceil(p*float64(len(durations)))) - 1
if rank < 0 {
rank = 0
}
if rank >= len(durations) {
rank = len(durations) - 1
}
return durations[rank].Milliseconds()
}
79 changes: 79 additions & 0 deletions core/http/middleware/trace_summary_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
// SPDX-License-Identifier: MIT

package middleware

import (
"time"

. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)

var _ = Describe("API trace summary", func() {
exchange := func(age time.Duration, status int, dur time.Duration) APIExchange {
return APIExchange{
Timestamp: time.Now().Add(-age),
Duration: dur,
Response: APIExchangeResponse{Status: status},
}
}

It("counts only what falls inside the window", func() {
traces := []APIExchange{
exchange(1*time.Hour, 200, 10*time.Millisecond),
exchange(2*time.Hour, 200, 10*time.Millisecond),
// Older than the window: must not be counted at all.
exchange(48*time.Hour, 500, 10*time.Millisecond),
}
s := summarize(traces, 24*time.Hour, 6)
Expect(s.Total).To(Equal(2))
Expect(s.Errors).To(BeZero())
})

It("treats 5xx and a transport error as failures, but not 4xx", func() {
traces := []APIExchange{
exchange(time.Minute, 500, time.Millisecond),
exchange(time.Minute, 503, time.Millisecond),
// A client sending a bad request is not the server failing.
exchange(time.Minute, 404, time.Millisecond),
exchange(time.Minute, 200, time.Millisecond),
}
traces[3].Error = "connection reset"

s := summarize(traces, 24*time.Hour, 6)
Expect(s.Total).To(Equal(4))
Expect(s.Errors).To(Equal(3))
})

It("reports p95 as a real percentile rather than the slowest request", func() {
traces := make([]APIExchange, 0, 100)
for i := 1; i <= 100; i++ {
traces = append(traces, exchange(time.Minute, 200, time.Duration(i)*time.Millisecond))
}
s := summarize(traces, 24*time.Hour, 6)
// 95th of 1..100ms, not the 100ms max.
Expect(s.P95Millis).To(BeNumerically("~", 95, 1))
})

It("buckets oldest-first so a sparkline reads left to right", func() {
traces := []APIExchange{
exchange(30*time.Minute, 200, time.Millisecond),
exchange(30*time.Minute, 200, time.Millisecond),
exchange(5*time.Hour, 200, time.Millisecond),
}
s := summarize(traces, 6*time.Hour, 6)
Expect(s.Buckets).To(HaveLen(6))
Expect(s.Buckets[0].Count).To(Equal(1), "the 5h-old request lands in the first bucket")
Expect(s.Buckets[5].Count).To(Equal(2), "the recent pair lands in the last")
})

It("returns an empty, non-nil summary when nothing has been traced", func() {
s := summarize(nil, 24*time.Hour, 6)
Expect(s.Total).To(BeZero())
Expect(s.Errors).To(BeZero())
Expect(s.P95Millis).To(BeZero())
// A nil slice serialises as null and breaks .map() in the browser.
Expect(s.Buckets).NotTo(BeNil())
Expect(s.Buckets).To(HaveLen(6))
})
})
4 changes: 3 additions & 1 deletion core/http/react-ui/e2e/admin-console.spec.js
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,9 @@ test.describe('Admin console', () => {
await page.goto('/app/backends')
const rail = page.locator('.console-rail')
await expect(rail).toBeVisible()
for (const group of ['Inference', 'Cluster', 'Observability', 'Access', 'System']) {
// Four groups since the overview landed: Inference folded into Runtime
// (both are "the runtime right now"), Access and System into Administration.
for (const group of ['Runtime', 'Cluster', 'Observability', 'Administration']) {
await expect(rail.locator('.console-group-title', { hasText: group })).toBeVisible()
}
})
Expand Down
55 changes: 55 additions & 0 deletions core/http/react-ui/e2e/chat-transcript.spec.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
import { test, expect } from './coverage-fixtures.js'

// Chat reads as a transcript rather than a bubble thread (mock 04).

const CHAT = {
chats: [{
id: 'c1', name: 'Transcript', model: 'mock-model',
history: [
{ role: 'user', content: 'Which backends do I have?' },
{ role: 'assistant', content: 'Seven are installed.' },
],
}],
activeChatId: 'c1',
}

test.describe('Chat transcript', () => {
test.beforeEach(async ({ page }) => {
await page.addInitScript(chat => {
localStorage.setItem('localai_chats_data', JSON.stringify(chat))
}, CHAT)
await page.goto('/app/chat')
})

test('neither role is a filled, rounded bubble', async ({ page }) => {
const user = page.locator('.chat-message-user .chat-message-content').first()
await expect(user).toBeVisible()
const cs = await user.evaluate(el => {
const s = getComputedStyle(el)
return { radius: s.borderTopLeftRadius, shadow: s.boxShadow }
})
// A rounded filled bubble carries the speaker in shape and side; a
// transcript carries it in words, which survives being read aloud.
expect(cs.radius).toBe('0px')
expect(cs.shadow).toBe('none')
})

test('both turns run full width in one column, not left and right', async ({ page }) => {
const user = page.locator('.chat-message-user').first()
const assistant = page.locator('.chat-message-assistant').first()
const [u, a] = [await user.boundingBox(), await assistant.boundingBox()]
expect(Math.abs(u.x - a.x)).toBeLessThan(2)
})

test('every turn says who is speaking', async ({ page }) => {
await expect(page.locator('.chat-message-user .chat-message-model')).toHaveText('You')
await expect(page.locator('.chat-message-assistant .chat-message-model').first())
.toHaveText('mock-model')
})

test('turns are separated by a rule', async ({ page }) => {
const border = await page.locator('.chat-message').first()
.evaluate(el => getComputedStyle(el).borderBottomStyle)
expect(border).toBe('solid')
})
})
68 changes: 68 additions & 0 deletions core/http/react-ui/e2e/console-narrow.spec.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,68 @@
import { test, expect } from './coverage-fixtures.js'

// Small-screen behaviour of the Operate console and the dashboard stat cards.
//
// Both defects here are about a narrow viewport but neither is only a narrow
// viewport problem: the stat cards were being laid out by the wrong rule at
// every width, and the rail's height was never bounded.

test.describe('Operate console on a narrow screen', () => {
test('expanding the rail leaves the page still on screen', async ({ page }) => {
await page.setViewportSize({ width: 390, height: 800 })
await page.goto('/app/manage')

const toggle = page.locator('.console-rail-toggle')
await expect(toggle).toBeVisible()
await toggle.click()
await expect(page.locator('.console-rail-groups')).toBeVisible()

// Thirteen destinations in one column is taller than a phone. If opening
// the menu pushes the page's own heading past the fold, the menu has
// replaced the page instead of annotating it.
// Manage titles itself with .view-bar__title rather than .page-title.
const heading = page.locator('.page-title, .view-bar__title').first()
const box = await heading.boundingBox()
expect(box).not.toBeNull()
expect(box.y).toBeLessThan(800)
})

test('the rail scrolls internally rather than growing without bound', async ({ page }) => {
await page.setViewportSize({ width: 390, height: 800 })
await page.goto('/app/manage')
await page.locator('.console-rail-toggle').click()

const groups = page.locator('.console-rail-groups')
await expect(groups).toBeVisible()
const height = await groups.evaluate(el => el.getBoundingClientRect().height)
expect(height).toBeLessThan(800)
})
})

test.describe('Dashboard stat cards', () => {
// Two components both claimed `.stat-grid`: the dashboard card strip and the
// detail-pane StatGrid added with the split views. The later rule won, so the
// cards were laid out on 120px columns with a 1px gap meant for something
// else, and their labels were clipped.
for (const width of [768, 1024]) {
test(`labels are not clipped at ${width}px`, async ({ page }) => {
await page.setViewportSize({ width, height: 1000 })
await page.goto('/app/manage')
const labels = page.locator('.stat-card__label')
await expect(labels.first()).toBeVisible()

const clipped = await labels.evaluateAll(els =>
els.filter(el => el.scrollWidth > el.clientWidth + 1).map(el => el.textContent))
expect(clipped).toEqual([])
})
}

test('cards keep the card gap, not the hairline gap of the detail pane grid', async ({ page }) => {
await page.setViewportSize({ width: 1024, height: 1000 })
await page.goto('/app/manage')
const strip = page.locator('.manage-summary')
await expect(strip).toBeVisible()
const gap = await strip.evaluate(el => parseFloat(getComputedStyle(el).columnGap))
// 1px is the detail-pane StatGrid's hairline; the card strip wants real space.
expect(gap).toBeGreaterThan(4)
})
})
Loading
Loading