lnp_ml/server_gpu_test.sh

109 lines
3.2 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
ROOT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
cd "${ROOT_DIR}"
COMPOSE_FILE=${COMPOSE_FILE:-docker-compose-gpu.yml}
API_PORT=${API_PORT:-18000}
API_URL=${API_URL:-http://127.0.0.1:${API_PORT}}
LLM_INFERENCE_BATCH_SIZE=${LLM_INFERENCE_BATCH_SIZE:-4}
export API_PORT
export LLM_INFERENCE_BATCH_SIZE
fail() {
echo "ERROR: $*" >&2
exit 1
}
assert_real_file() {
local path=$1
[ -f "${path}" ] || fail "missing file: ${path}"
if head -n 1 "${path}" 2>/dev/null | grep -q "git-lfs.github.com/spec"; then
fail "${path} is a Git LFS pointer, not the real artifact"
fi
}
preflight() {
command -v docker >/dev/null || fail "docker is not installed"
command -v nvidia-smi >/dev/null || fail "nvidia-smi is not installed"
nvidia-smi >/dev/null || fail "NVIDIA driver/GPU is unavailable"
docker compose version >/dev/null || fail "docker compose plugin is unavailable"
assert_real_file models/final/model.pt
assert_real_file data/interim/internal.csv
assert_real_file models/qwen2.5-7b-instruct/config.json
local fold
for fold in 0 1 2 3 4; do
assert_real_file "models/mpnn/all_amine_split_for_LiON/cv_${fold}/fold_0/model_0/model.pt"
done
find models/qwen2.5-7b-instruct -maxdepth 1 -type f \
\( -name '*.safetensors' -o -name '*.bin' \) -print -quit | grep -q . \
|| fail "Qwen weight shards (*.safetensors or *.bin) are missing"
docker compose -f "${COMPOSE_FILE}" config >/dev/null
echo "preflight: OK"
}
wait_for_api() {
local attempt
for attempt in $(seq 1 120); do
if curl -fsS "${API_URL}/" >/dev/null 2>&1; then
echo "API ready after ${attempt} checks"
return 0
fi
sleep 2
done
docker compose -f "${COMPOSE_FILE}" logs --tail=200 api || true
fail "API did not become healthy within 240 seconds"
}
start() {
preflight
docker compose -f "${COMPOSE_FILE}" build
docker compose -f "${COMPOSE_FILE}" up -d
wait_for_api
docker compose -f "${COMPOSE_FILE}" ps
curl -fsS "${API_URL}/"
echo
}
request_item() {
printf '%s' '{"smiles":"CC(C)NCCNC(C)C","cationic_lipid_to_mrna_ratio":10.0,"cationic_lipid_mol_ratio":35.0,"phospholipid_mol_ratio":16.0,"cholesterol_mol_ratio":46.5,"peg_lipid_mol_ratio":2.5,"helper_lipid":"DOPE","route":"intravenous"}'
}
smoke() {
wait_for_api
local item payload i
item=$(request_item)
curl -fsS -H 'Content-Type: application/json' \
-d "${item}" "${API_URL}/predict" >/tmp/lnp-single-response.json
echo "single prediction: OK"
payload='{"items":['
for i in $(seq 1 32); do
[ "${i}" -eq 1 ] || payload+=','
payload+="${item}"
done
payload+='],"batch_size":32,"use_llm":true}'
curl -fsS -H 'Content-Type: application/json' \
-d "${payload}" "${API_URL}/predict/batch" >/tmp/lnp-batch-response.json
grep -q '"n_succeeded":32' /tmp/lnp-batch-response.json \
|| fail "batch response did not report 32 successes"
echo "32-item LLM prediction: OK"
docker compose -f "${COMPOSE_FILE}" logs --tail=200 api \
| grep -E 'LLM inference micro-batching|ERROR|CUDA out of memory' || true
}
case "${1:-}" in
preflight) preflight ;;
start) start ;;
smoke) smoke ;;
*) echo "Usage: $0 {preflight|start|smoke}" >&2; exit 2 ;;
esac