| 1 | //! Liveness and readiness. | |
| 2 | //! | |
| 3 | //! `/healthz` answers whether the process is up; `/readyz` answers whether it | |
| 4 | //! can serve, which means the database is reachable and the pool can still hand | |
| 5 | //! out a connection. | |
| 6 | //! | |
| 7 | //! The container healthcheck reads `/readyz`. It used to read `/healthz` on the | |
| 8 | //! reasoning that restarting on readiness turns a transient database blip into a | |
| 9 | //! restart loop — but nothing restarts on either signal here (compose restarts | |
| 10 | //! on exit), so all that bought was a container reporting healthy while it | |
| 11 | //! served nothing but 500s. | |
| 12 | ||
| 13 | use axum::extract::State; | |
| 14 | use axum::http::StatusCode; | |
| 15 | use axum::response::IntoResponse; | |
| 16 | ||
| 17 | use crate::state::AppState; | |
| 18 | ||
| 19 | pub async fn healthz() -> impl IntoResponse { | |
| 20 | (StatusCode::OK, "ok\n") | |
| 21 | } | |
| 22 | ||
| 23 | pub async fn readyz(State(state): State<AppState>) -> impl IntoResponse { | |
| 24 | // Reported either way: a pool that is at its ceiling with nothing idle is | |
| 25 | // the signature of connections stuck in the pool rather than of a database | |
| 26 | // that has gone away, and the two need different responses from whoever is | |
| 27 | // reading this. | |
| 28 | let open = state.db.size(); | |
| 29 | let idle = state.db.num_idle(); | |
| 30 | ||
| 31 | match df_db::ping(&state.db).await { | |
| 32 | Ok(()) => ( | |
| 33 | StatusCode::OK, | |
| 34 | format!("ready\npool: {idle} idle / {open} open\n"), | |
| 35 | ), | |
| 36 | Err(e) => { | |
| 37 | tracing::error!( | |
| 38 | pool_open = open, | |
| 39 | pool_idle = idle, | |
| 40 | "readiness check failed: {e}" | |
| 41 | ); | |
| 42 | ( | |
| 43 | StatusCode::SERVICE_UNAVAILABLE, | |
| 44 | format!("database unreachable\npool: {idle} idle / {open} open\n"), | |
| 45 | ) | |
| 46 | } | |
| 47 | } | |
| 48 | } |
48 lines · Rust