//! A JSON extractor that fails with the status codes §4 specifies. //! //! Axum's own `Json` rejects a body that parses as JSON but does not match the //! target type with **422 Unprocessable Entity**. §4 is explicit that this case //! is **`400`** — "malformed, or contains an unrecognised or forbidden field" — //! and that distinction is load-bearing: §6 requires that a client which forgets //! to strip `movie` or `jellyfin_id` gets "a hard `400` naming the offending //! field". A client checking for 400 would mishandle a 422. //! //! This wrapper also guarantees the field name reaches the caller, since serde's //! `deny_unknown_fields` error text is what identifies the offending key. use axum::extract::{FromRequest, Request}; use axum::http::header::CONTENT_TYPE; use crate::error::ApiError; /// Drop-in replacement for `axum::Json` on request bodies. pub struct Json(pub T); impl FromRequest for Json where T: serde::de::DeserializeOwned, S: Send + Sync, { type Rejection = ApiError; async fn from_request(req: Request, state: &S) -> Result { // A wrong content type is the client's mistake, reported as such rather // than as a parse failure. let content_type = req.headers().get(CONTENT_TYPE).and_then(|v| v.to_str().ok()).unwrap_or("").to_string(); let mime = content_type.split(';').next().unwrap_or("").trim().to_ascii_lowercase(); if !(mime == "application/json" || mime.ends_with("+json")) { return Err(ApiError::BadRequest("expected content-type: application/json".into())); } // A declared charset other than UTF-8 is refused up front, so the client // learns what is wrong rather than receiving a confusing parse error from // deep inside the document. See `require_utf8` for why UTF-8 is the only // accepted encoding. if let Some(charset) = content_type.split(';').skip(1).filter_map(|p| p.trim().strip_prefix("charset=")).next() { let charset = charset.trim().trim_matches('"').to_ascii_lowercase(); if !matches!(charset.as_str(), "utf-8" | "utf8") { return Err(ApiError::BadRequest(format!( "unsupported charset {charset:?}: JSON must be UTF-8 encoded (RFC 8259 §8.1)" ))); } } let bytes = axum::body::Bytes::from_request(req, state).await.map_err(|e| { // §6 stage 1: the body cap aborts mid-transfer, and that must surface // as `413`, not as a generic parse error. Axum folds the length-limit // case into `FailedToBufferBody`, so the status it chose is the // reliable discriminator. if e.status() == axum::http::StatusCode::PAYLOAD_TOO_LARGE { ApiError::PayloadTooLarge("request body exceeds the limit for this route".into()) } else { ApiError::BadRequest(format!("could not read request body: {e}")) } })?; // Encoding is checked before parsing, so a mis-encoded body gets an // actionable message instead of whatever the parser happens to trip over. let text = require_utf8(&bytes)?; serde_json::from_str(text) .map(Json) // serde's message names the offending field, which is exactly what §6 // requires the response to identify. .map_err(|e| ApiError::BadRequest(e.to_string())) } } /// Enforces that the body is UTF-8, naming the encoding it appears to be. /// /// **UTF-8 is the only accepted encoding, deliberately.** RFC 8259 §8.1 requires /// it for JSON exchanged outside a closed ecosystem, and this is a public, /// federated API. Three further reasons make it the right call *here* /// specifically, rather than merely conventional: /// /// 1. **§9a content addressing hashes bytes.** `content_id` is a SHA-256 over the /// canonical form, so the same manifest submitted in two encodings would /// produce two different ids — silently defeating federation deduplication. /// That is precisely the failure mode §9a quantises scene times to avoid, and /// it would be reintroduced at the encoding layer. /// 2. **UTF-16 admits lone surrogates**, which have no UTF-8 representation. A /// field able to carry them is a channel for bytes that survive validation but /// are not text — against §5a's premise that no field can carry a payload. /// 3. **§5a's character class assumes well-formed Unicode scalar values.** NFC /// normalisation and the category checks are defined over scalars, so admitting /// an encoding that can express non-scalars would undermine both. /// /// serde_json would reject non-UTF-8 anyway; the value added here is a diagnosable /// error rather than a misleading one. A UTF-16 body otherwise fails with "key /// must be a string", which points an operator at the wrong problem entirely. /// TRACES: DR-010, DR-013 | SR-003 fn require_utf8(bytes: &[u8]) -> Result<&str, ApiError> { // A BOM is not valid JSON (RFC 8259 §8.1: "implementations MUST NOT add a // byte order mark"), and it is the clearest signal of an encoding mistake, so // it is named rather than left to the parser. let encoding_hint = match bytes { [0xEF, 0xBB, 0xBF, ..] => Some("UTF-8 with a byte order mark"), [0xFF, 0xFE, 0x00, 0x00, ..] => Some("UTF-32LE"), [0x00, 0x00, 0xFE, 0xFF, ..] => Some("UTF-32BE"), [0xFF, 0xFE, ..] => Some("UTF-16LE"), [0xFE, 0xFF, ..] => Some("UTF-16BE"), // Unmarked UTF-16 is the common case, since encoders often omit the BOM. // A JSON document always begins with an ASCII character, so an // interleaved NUL in the first two bytes is conclusive. [0x00, b, ..] if b.is_ascii_graphic() => Some("UTF-16BE (no BOM)"), [b, 0x00, ..] if b.is_ascii_graphic() => Some("UTF-16LE (no BOM)"), _ => None, }; if let Some(encoding) = encoding_hint { return Err(ApiError::BadRequest(format!( "request body appears to be {encoding}: JSON must be UTF-8 encoded \ without a byte order mark (RFC 8259 §8.1)" ))); } std::str::from_utf8(bytes).map_err(|e| { ApiError::BadRequest(format!( "request body is not valid UTF-8 at byte {}: JSON must be UTF-8 encoded \ (RFC 8259 §8.1)", e.valid_up_to() )) }) } #[cfg(test)] mod tests { use super::*; use axum::http::StatusCode; use axum::response::IntoResponse; #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] struct Probe { _wanted: i64, } async fn extract(body: &'static str, content_type: Option<&str>) -> StatusCode { let mut builder = Request::builder().method("POST").uri("/"); if let Some(ct) = content_type { builder = builder.header(CONTENT_TYPE, ct); } let req = builder.body(axum::body::Body::from(body)).unwrap(); match Json::::from_request(req, &()).await { Ok(_) => StatusCode::OK, Err(e) => e.into_response().status(), } } #[tokio::test] async fn schema_mismatch_is_400_not_422() { // The whole reason this extractor exists (§4, §6). assert_eq!( extract(r#"{"unexpected":1}"#, Some("application/json")).await, StatusCode::BAD_REQUEST ); } #[tokio::test] async fn malformed_json_is_400() { assert_eq!(extract("{ nope", Some("application/json")).await, StatusCode::BAD_REQUEST); } #[tokio::test] async fn missing_content_type_is_400() { assert_eq!(extract(r#"{"_wanted":1}"#, None).await, StatusCode::BAD_REQUEST); } #[tokio::test] async fn content_type_parameters_are_tolerated() { assert_eq!( extract(r#"{"_wanted":1}"#, Some("application/json; charset=utf-8")).await, StatusCode::OK ); } #[tokio::test] async fn valid_body_extracts() { assert_eq!(extract(r#"{"_wanted":1}"#, Some("application/json")).await, StatusCode::OK); } #[tokio::test] async fn an_explicit_utf8_charset_is_accepted() { for ct in [ "application/json; charset=utf-8", "application/json;charset=UTF-8", "application/json; charset=\"utf-8\"", "application/json; charset=utf8", ] { assert_eq!(extract(r#"{"_wanted":1}"#, Some(ct)).await, StatusCode::OK, "{ct}"); } } #[tokio::test] async fn a_non_utf8_charset_is_refused_by_name() { for ct in [ "application/json; charset=utf-16", "application/json; charset=iso-8859-1", "application/json; charset=windows-1252", ] { assert_eq!( extract(r#"{"_wanted":1}"#, Some(ct)).await, StatusCode::BAD_REQUEST, "{ct}" ); } } /// Builds a request from raw bytes, since these bodies are not valid `&str`. async fn extract_bytes(body: Vec) -> Result<(), ApiError> { let req = Request::builder() .method("POST") .uri("/") .header(CONTENT_TYPE, "application/json") .body(axum::body::Body::from(body)) .unwrap(); Json::::from_request(req, &()).await.map(|_| ()) } #[tokio::test] async fn utf16_bodies_are_rejected_with_an_actionable_message() { // The reason this check exists: serde_json rejects UTF-16 anyway, but with // "key must be a string", which points an operator at the wrong problem. let doc = r#"{"_wanted":1}"#; let le: Vec = doc.encode_utf16().flat_map(|u| u.to_le_bytes()).collect(); let err = extract_bytes(le).await.unwrap_err().to_string(); assert!(err.contains("UTF-16LE"), "should name the encoding: {err}"); assert!(err.contains("UTF-8"), "should say what is required: {err}"); let be: Vec = doc.encode_utf16().flat_map(|u| u.to_be_bytes()).collect(); let err = extract_bytes(be).await.unwrap_err().to_string(); assert!(err.contains("UTF-16BE"), "should name the encoding: {err}"); // With BOMs. let mut le_bom = vec![0xFF, 0xFE]; le_bom.extend(doc.encode_utf16().flat_map(|u| u.to_le_bytes())); assert!(extract_bytes(le_bom).await.is_err()); let mut be_bom = vec![0xFE, 0xFF]; be_bom.extend(doc.encode_utf16().flat_map(|u| u.to_be_bytes())); assert!(extract_bytes(be_bom).await.is_err()); } #[tokio::test] async fn a_utf8_bom_is_rejected() { // RFC 8259 §8.1: implementations MUST NOT add a byte order mark. let mut body = vec![0xEF, 0xBB, 0xBF]; body.extend_from_slice(br#"{"_wanted":1}"#); let err = extract_bytes(body).await.unwrap_err().to_string(); assert!(err.contains("byte order mark"), "{err}"); } #[tokio::test] async fn invalid_utf8_is_rejected_with_the_offending_offset() { // A truncated multi-byte sequence inside an otherwise well-formed document. let body = b"{\"_wanted\":\"\xC3\x28\"}".to_vec(); let err = extract_bytes(body).await.unwrap_err().to_string(); assert!(err.contains("not valid UTF-8"), "{err}"); assert!(err.contains("byte 12"), "should locate the failure: {err}"); } #[tokio::test] async fn valid_multibyte_utf8_is_accepted() { // The check must not reject legitimate non-ASCII content — actor names are // routinely non-Latin (§5a accepts any Unicode letter). // Rejected for the unknown `_note` field, not for its encoding — which is // the distinction being asserted. let body = r#"{"_wanted":1,"_note":"宮崎 駿 Renée"}"#.as_bytes().to_vec(); let err = extract_bytes(body).await.unwrap_err().to_string(); assert!(err.contains("_note"), "should fail on the schema, not the encoding: {err}"); } #[tokio::test] async fn an_empty_body_is_not_mistaken_for_an_encoding_problem() { let err = extract_bytes(Vec::new()).await.unwrap_err().to_string(); assert!(!err.contains("UTF-16"), "empty body is a parse error, not an encoding one: {err}"); } }