diff --git a/README.md b/README.md index d8a38ff..a0909ce 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,23 @@ OpenAI-compatible LLM inference server using `llama-cpp-2` (Rust). ### `GET /health` ```json -{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking"} +{"status": "ok", "model": "minicpm5-1b-fable5-v2-thinking", "uptime_s": 1234, "n_ctx": 8192, "version": "0.1.0"} +``` + +### `GET /metrics` +Prometheus text exposition (no auth) — request counters, token usage, generation +latency/throughput, process uptime: + +``` +llm_api_requests_total # total /v1/chat/completions +llm_api_errors_total # errored requests +llm_api_streaming_requests_total # stream: true requests +llm_api_aborted_requests_total # aborted generations (client disconnect) +llm_api_prompt_tokens_total # prompt tokens accepted +llm_api_completion_tokens_total # tokens generated +llm_api_generation_ms_total # generation time (ms) +llm_api_tokens_per_second # lifetime throughput gauge +llm_api_build_info{version,model} # identity ``` ### `GET /v1/models` @@ -53,6 +69,7 @@ MODEL_PATH=/path/to/model.gguf ./target/release/llm-api | `SERVER_PORT` | `4010` | Listen port | | `RUST_LOG` | `info` | Log level | | `N_CTX` / `N_BATCH` / `N_THREADS` | `8192` / `512` / `4` | llama.cpp context/batch/threads | +| `MAX_TOKENS` | `2048` | Hard cap untuk `max_tokens` request (0 = unlimited) | ### Smoke test (setelah deploy) diff --git a/scripts/smoke-test.sh b/scripts/smoke-test.sh index aa6d28e..141dbaa 100755 --- a/scripts/smoke-test.sh +++ b/scripts/smoke-test.sh @@ -32,16 +32,25 @@ check() { fi } -echo "== 1. Health ==" +echo "== 1. Health ===" health=$(curl -sf "$BASE_URL/health") || { echo "FAIL: /health unreachable"; exit 1; } echo "$health" | jq -e ".status == \"ok\" and .model == \"$MODEL_ID\"" >/dev/null check "health melaporkan $MODEL_ID" $? +echo "$health" | jq -e '.uptime_s >= 0 and (.version | type == "string")' >/dev/null +check "health punya uptime + version" $? echo "== 2. Models ==" models=$(curl -sf "$BASE_URL/v1/models") echo "$models" | jq -e ".data[0].id == \"$MODEL_ID\"" >/dev/null check "models mencantumkan $MODEL_ID" $? +echo "== 2b. Metrics ==" +metrics=$(curl -sf "$BASE_URL/metrics") || { echo "FAIL: /metrics unreachable"; exit 1; } +echo "$metrics" | grep -q "llm_api_requests_total" +check "metrics punya llm_api_requests_total" $? +echo "$metrics" | grep -q "llm_api_build_info{" +check "metrics punya llm_api_build_info" $? + echo "== 3. Chat non-streaming ==" resp=$(curl -sf "${AUTH[@]}" -H "Content-Type: application/json" \ -d "{\"model\":\"$MODEL_ID\",\"messages\":[{\"role\":\"user\",\"content\":\"Say hi\"}],\"max_tokens\":64}" \ @@ -50,6 +59,8 @@ echo "$resp" | jq -e '.choices[0].message.content | type == "string"' >/dev/null check "non-streaming mengembalikan content" $? echo "$resp" | jq -e '.usage.total_tokens > 0' >/dev/null check "non-streaming berisi usage" $? +echo "$resp" | jq -e '.usage.duration_ms > 0 and .usage.tokens_per_second > 0' >/dev/null +check "non-streaming berisi timing (duration_ms + tok/s)" $? echo "== 4. Chat streaming ==" stream=$(curl -sfN "${AUTH[@]}" -H "Content-Type: application/json" \