docs: README /metrics + MAX_TOKENS, smoke-test health/metrics/timing checks
This commit is contained in:
@@ -10,7 +10,23 @@ OpenAI-compatible LLM inference server using `llama-cpp-2` (Rust).
|
|||||||
|
|
||||||
### `GET /health`
|
### `GET /health`
|
||||||
```json
|
```json
|
||||||
{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking"}
|
{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking", "uptime_s": 1234, "n_ctx": 8192, "version": "0.1.0"}
|
||||||
|
```
|
||||||
|
|
||||||
|
### `GET /metrics`
|
||||||
|
Prometheus text exposition (no auth) — request counters, token usage, generation
|
||||||
|
latency/throughput, process uptime:
|
||||||
|
|
||||||
|
```
|
||||||
|
llm_api_requests_total # total /v1/chat/completions
|
||||||
|
llm_api_errors_total # errored requests
|
||||||
|
llm_api_streaming_requests_total # stream: true requests
|
||||||
|
llm_api_aborted_requests_total # aborted generations (client disconnect)
|
||||||
|
llm_api_prompt_tokens_total # prompt tokens accepted
|
||||||
|
llm_api_completion_tokens_total # tokens generated
|
||||||
|
llm_api_generation_ms_total # generation time (ms)
|
||||||
|
llm_api_tokens_per_second # lifetime throughput gauge
|
||||||
|
llm_api_build_info{version,model} # identity
|
||||||
```
|
```
|
||||||
|
|
||||||
### `GET /v1/models`
|
### `GET /v1/models`
|
||||||
@@ -53,6 +69,7 @@ MODEL_PATH=/path/to/model.gguf ./target/release/llm-api
|
|||||||
| `SERVER_PORT` | `4010` | Listen port |
|
| `SERVER_PORT` | `4010` | Listen port |
|
||||||
| `RUST_LOG` | `info` | Log level |
|
| `RUST_LOG` | `info` | Log level |
|
||||||
| `N_CTX` / `N_BATCH` / `N_THREADS` | `8192` / `512` / `4` | llama.cpp context/batch/threads |
|
| `N_CTX` / `N_BATCH` / `N_THREADS` | `8192` / `512` / `4` | llama.cpp context/batch/threads |
|
||||||
|
| `MAX_TOKENS` | `2048` | Hard cap untuk `max_tokens` request (0 = unlimited) |
|
||||||
|
|
||||||
### Smoke test (setelah deploy)
|
### Smoke test (setelah deploy)
|
||||||
|
|
||||||
|
|||||||
+12
-1
@@ -32,16 +32,25 @@ check() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
echo "== 1. Health =="
|
echo "== 1. Health ==="
|
||||||
health=$(curl -sf "$BASE_URL/health") || { echo "FAIL: /health unreachable"; exit 1; }
|
health=$(curl -sf "$BASE_URL/health") || { echo "FAIL: /health unreachable"; exit 1; }
|
||||||
echo "$health" | jq -e ".status == \"ok\" and .model == \"$MODEL_ID\"" >/dev/null
|
echo "$health" | jq -e ".status == \"ok\" and .model == \"$MODEL_ID\"" >/dev/null
|
||||||
check "health melaporkan $MODEL_ID" $?
|
check "health melaporkan $MODEL_ID" $?
|
||||||
|
echo "$health" | jq -e '.uptime_s >= 0 and (.version | type == "string")' >/dev/null
|
||||||
|
check "health punya uptime + version" $?
|
||||||
|
|
||||||
echo "== 2. Models =="
|
echo "== 2. Models =="
|
||||||
models=$(curl -sf "$BASE_URL/v1/models")
|
models=$(curl -sf "$BASE_URL/v1/models")
|
||||||
echo "$models" | jq -e ".data[0].id == \"$MODEL_ID\"" >/dev/null
|
echo "$models" | jq -e ".data[0].id == \"$MODEL_ID\"" >/dev/null
|
||||||
check "models mencantumkan $MODEL_ID" $?
|
check "models mencantumkan $MODEL_ID" $?
|
||||||
|
|
||||||
|
echo "== 2b. Metrics =="
|
||||||
|
metrics=$(curl -sf "$BASE_URL/metrics") || { echo "FAIL: /metrics unreachable"; exit 1; }
|
||||||
|
echo "$metrics" | grep -q "llm_api_requests_total"
|
||||||
|
check "metrics punya llm_api_requests_total" $?
|
||||||
|
echo "$metrics" | grep -q "llm_api_build_info{"
|
||||||
|
check "metrics punya llm_api_build_info" $?
|
||||||
|
|
||||||
echo "== 3. Chat non-streaming =="
|
echo "== 3. Chat non-streaming =="
|
||||||
resp=$(curl -sf "${AUTH[@]}" -H "Content-Type: application/json" \
|
resp=$(curl -sf "${AUTH[@]}" -H "Content-Type: application/json" \
|
||||||
-d "{\"model\":\"$MODEL_ID\",\"messages\":[{\"role\":\"user\",\"content\":\"Say hi\"}],\"max_tokens\":64}" \
|
-d "{\"model\":\"$MODEL_ID\",\"messages\":[{\"role\":\"user\",\"content\":\"Say hi\"}],\"max_tokens\":64}" \
|
||||||
@@ -50,6 +59,8 @@ echo "$resp" | jq -e '.choices[0].message.content | type == "string"' >/dev/null
|
|||||||
check "non-streaming mengembalikan content" $?
|
check "non-streaming mengembalikan content" $?
|
||||||
echo "$resp" | jq -e '.usage.total_tokens > 0' >/dev/null
|
echo "$resp" | jq -e '.usage.total_tokens > 0' >/dev/null
|
||||||
check "non-streaming berisi usage" $?
|
check "non-streaming berisi usage" $?
|
||||||
|
echo "$resp" | jq -e '.usage.duration_ms > 0 and .usage.tokens_per_second > 0' >/dev/null
|
||||||
|
check "non-streaming berisi timing (duration_ms + tok/s)" $?
|
||||||
|
|
||||||
echo "== 4. Chat streaming =="
|
echo "== 4. Chat streaming =="
|
||||||
stream=$(curl -sfN "${AUTH[@]}" -H "Content-Type: application/json" \
|
stream=$(curl -sfN "${AUTH[@]}" -H "Content-Type: application/json" \
|
||||||
|
|||||||
Reference in New Issue
Block a user