docs: README /metrics + MAX_TOKENS, smoke-test health/metrics/timing checks

This commit is contained in:
asepharyana
2026-08-03 11:44:27 +07:00
parent e8f90fc9b1
commit 5f7ead5503
2 changed files with 30 additions and 2 deletions
+18 -1
View File
@@ -10,7 +10,23 @@ OpenAI-compatible LLM inference server using `llama-cpp-2` (Rust).
### `GET /health`
```json
{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking"}
{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking", "uptime_s": 1234, "n_ctx": 8192, "version": "0.1.0"}
```
### `GET /metrics`
Prometheus text exposition (no auth) — request counters, token usage, generation
latency/throughput, process uptime:
```
llm_api_requests_total # total /v1/chat/completions
llm_api_errors_total # errored requests
llm_api_streaming_requests_total # stream: true requests
llm_api_aborted_requests_total # aborted generations (client disconnect)
llm_api_prompt_tokens_total # prompt tokens accepted
llm_api_completion_tokens_total # tokens generated
llm_api_generation_ms_total # generation time (ms)
llm_api_tokens_per_second # lifetime throughput gauge
llm_api_build_info{version,model} # identity
```
### `GET /v1/models`
@@ -53,6 +69,7 @@ MODEL_PATH=/path/to/model.gguf ./target/release/llm-api
| `SERVER_PORT` | `4010` | Listen port |
| `RUST_LOG` | `info` | Log level |
| `N_CTX` / `N_BATCH` / `N_THREADS` | `8192` / `512` / `4` | llama.cpp context/batch/threads |
| `MAX_TOKENS` | `2048` | Hard cap untuk `max_tokens` request (0 = unlimited) |
### Smoke test (setelah deploy)
+12 -1
View File
@@ -32,16 +32,25 @@ check() {
fi
}
echo "== 1. Health =="
echo "== 1. Health ==="
health=$(curl -sf "$BASE_URL/health") || { echo "FAIL: /health unreachable"; exit 1; }
echo "$health" | jq -e ".status == \"ok\" and .model == \"$MODEL_ID\"" >/dev/null
check "health melaporkan $MODEL_ID" $?
echo "$health" | jq -e '.uptime_s >= 0 and (.version | type == "string")' >/dev/null
check "health punya uptime + version" $?
echo "== 2. Models =="
models=$(curl -sf "$BASE_URL/v1/models")
echo "$models" | jq -e ".data[0].id == \"$MODEL_ID\"" >/dev/null
check "models mencantumkan $MODEL_ID" $?
echo "== 2b. Metrics =="
metrics=$(curl -sf "$BASE_URL/metrics") || { echo "FAIL: /metrics unreachable"; exit 1; }
echo "$metrics" | grep -q "llm_api_requests_total"
check "metrics punya llm_api_requests_total" $?
echo "$metrics" | grep -q "llm_api_build_info{"
check "metrics punya llm_api_build_info" $?
echo "== 3. Chat non-streaming =="
resp=$(curl -sf "${AUTH[@]}" -H "Content-Type: application/json" \
-d "{\"model\":\"$MODEL_ID\",\"messages\":[{\"role\":\"user\",\"content\":\"Say hi\"}],\"max_tokens\":64}" \
@@ -50,6 +59,8 @@ echo "$resp" | jq -e '.choices[0].message.content | type == "string"' >/dev/null
check "non-streaming mengembalikan content" $?
echo "$resp" | jq -e '.usage.total_tokens > 0' >/dev/null
check "non-streaming berisi usage" $?
echo "$resp" | jq -e '.usage.duration_ms > 0 and .usage.tokens_per_second > 0' >/dev/null
check "non-streaming berisi timing (duration_ms + tok/s)" $?
echo "== 4. Chat streaming =="
stream=$(curl -sfN "${AUTH[@]}" -H "Content-Type: application/json" \