//! TRACES: NFR-SEC-2 //! Taking the credentials back out of a line that already contains them. //! //! NFR-SEC-2 says credentials are never written to logs. NFR-OPS-1 restates it //! as an obligation on the diagnostics path specifically — "automatic //! redaction of credentials and tokens" — and the word doing the work is //! *automatic*. A rule that every call site must remember is not a rule; it is //! a hope, and it fails the first time somebody debugging a 401 writes //! `log::debug!("{response:?}")` at two in the morning and forgets to take it //! out. So the redaction happens once, at the sink, on the formatted line, //! after every call site has had its say. //! //! On Android this is not a defence-in-depth nicety. The log lives in the //! app's *external* directory so that `adb pull` can reach it //! ([`crate::state`]), which means anyone holding the device can read it, and //! a Nextcloud app password in there is a working credential for the user's //! whole server. //! //! # What it looks for, and why not a regex //! //! A credential in a log line is nearly always adjacent to a word that names //! it — `appPassword`, `Authorization`, `token=` — because it got there //! through a `Debug` impl, a serialised request, or a URL. That adjacency is //! the only reliable signal there is: the app password Nextcloud issues is an //! opaque 72-character string with no structure to match on, so a scanner that //! hunted for *secret-shaped* text would find every base64 blob and every //! content hash in the file and redact those instead. //! //! Hence a keyword scanner, hand-rolled rather than `regex`. The workspace has //! no regex dependency and this is not worth acquiring one for — the grammar //! is "a known word, an `=` or a `:`, a value" plus the two forms that carry a //! credential with no keyword at all: an `Authorization` scheme and a URL's //! userinfo. Three rules, each of which fits in a function and has a test. //! //! # What it deliberately leaves alone //! //! **Filesystem paths, and the names of the user's own photographs.** They are //! not in NFR-SEC-2's list, they are not in NFR-SEC-5's, and they are the //! single most useful thing in a log about a file that would not open. A //! diagnostics file that says "failed to decode " is not a //! diagnostic. The rule NFR-OPS-1 states — an explicit preview-and-consent //! step before anything leaves the device — is the one that governs paths, and //! it governs them better than scrubbing would, because it lets the user look. //! //! **Face data (NFR-SEC-5).** Not because it is permitted here, but because //! this is the wrong instrument for it. An embedding is 512 floats; by the //! time one has reached a formatted log line, redacting it would be a guess at //! what a `[f32]` looks like in prose. NFR-SEC-5 makes confinement //! *structural* — "the code path does not exist" — and the sink's contribution //! to that is to not be one: it opens no bundle, uploads nothing, and adds no //! new route off the device. The one thing it does do about the shape of the //! problem is cap the line length (see [`super::MAX_LINE_BYTES`]), so a //! `{embedding:?}` that should never have been written costs kilobytes rather //! than megabytes and is obvious in the file rather than buried. /// What a redacted value is replaced by. Chosen to be greppable and obviously /// not a value: a reader finding this in a log knows something was removed, /// which an empty string or a row of asterisks would not tell them. pub const PLACEHOLDER: &str = ""; /// The words that introduce a credential. /// /// Matched case-insensitively and only on a whole-word boundary, so `password` /// does not fire inside `passwordless` and `token` does not fire inside /// `tokenizer`. Longer entries are not redundant with shorter ones for the /// same reason — `apppassword` has no boundary before its `password`, so /// without its own entry `"appPassword": "…"` would sail through, and that is /// the exact key Nextcloud's Login Flow v2 returns the credential under /// (`dr_sync_nextcloud::auth::AppCredentials`). const SENSITIVE_KEYS: &[&str] = &[ "apppassword", "app_password", "password", "passwd", "pwd", "access_token", "refresh_token", "id_token", "apikey", "api_key", "authorization", "credentials", "credential", "secret", "token", ]; /// The `Authorization` schemes that carry the credential in the value rather /// than after a keyword. `Basic dXNlcjpwdw==` is `login:password` in base64 — /// reversible with one shell command — and every WebDAV request this app makes /// is signed with one (`dr_sync_nextcloud`, `basic_auth`). const AUTH_SCHEMES: &[&str] = &["basic ", "bearer "]; /// Remove anything that looks like a credential from one formatted line. /// /// Cheap enough for the write path: one ASCII-lowercased copy and a single /// left-to-right pass, with no allocation per match beyond the output string. /// Non-ASCII is untouched by the lowercasing, which is what keeps the byte /// offsets of the copy and the original identical — `to_lowercase` would not, /// and a one-byte drift would slice a message mid-character. pub fn redact(line: &str) -> String { let lower = line.to_ascii_lowercase(); let low = lower.as_bytes(); let mut out = String::with_capacity(line.len()); // Everything from here to the current position is verbatim text not yet // copied, so an unmatched line copies itself in one `push_str`. let mut copied = 0usize; let mut i = 0usize; while i < low.len() { match secret_at(low, i) { Some((start, end)) => { out.push_str(&line[copied..start]); out.push_str(PLACEHOLDER); copied = end; i = end; } None => i += 1, } } out.push_str(&line[copied..]); out } /// The byte range of a secret beginning at or just after `i`, if there is one. /// /// Every rule is anchored on an ASCII byte, and every boundary it returns is /// found by scanning for ASCII delimiters — which cannot occur inside a /// multi-byte UTF-8 sequence — so the ranges are always char boundaries even /// though the scan is over bytes. fn secret_at(low: &[u8], i: usize) -> Option<(usize, usize)> { if let Some(range) = keyed_value_at(low, i) { return Some(range); } if let Some(range) = bare_scheme_at(low, i) { return Some(range); } url_userinfo_at(low, i) } /// `password=hunter2`, `"appPassword": "…"`, `Authorization: Basic …`. fn keyed_value_at(low: &[u8], i: usize) -> Option<(usize, usize)> { if i > 0 && is_word_byte(low[i - 1]) { return None; } for key in SENSITIVE_KEYS { if !low[i..].starts_with(key.as_bytes()) { continue; } let after = i + key.len(); // A whole word, so `tokenizer` is not a token and `secretary` is not // a secret. `continue` rather than `return`: a short key failing its // boundary test must not mask a longer one that would have passed, // which is how `credential` would otherwise swallow `credentials`. if low.get(after).is_some_and(|b| is_word_byte(*b)) { continue; } return value_after_assignment(low, after); } None } /// The value introduced by an `=` or a `:` following a key. /// /// Requiring one of those two is what keeps prose out of it. Plenty of log /// lines say the word — "password rejected", "no token yet" — and a rule that /// redacted the next word after any mention of a credential would make those /// lines unreadable while protecting nothing. fn value_after_assignment(low: &[u8], mut j: usize) -> Option<(usize, usize)> { j = skip_spaces(low, j); // The closing quote of a quoted key: `"password": "…"`. if matches!(low.get(j), Some(b'"' | b'\'')) { j = skip_spaces(low, j + 1); } if !matches!(low.get(j), Some(b'=' | b':')) { return None; } j = skip_spaces(low, j + 1); // An opening quote is stepped over rather than redacted, so the result is // still shaped like what it replaced: `"password": ""` reads as // a redacted field, `"password": ` reads as damage. let quote = match low.get(j) { Some(&q) if q == b'"' || q == b'\'' => { j += 1; Some(q) } _ => None, }; // `Authorization: Basic dXNlcjpwdw==` — keep the scheme, take the rest. // Which scheme was in use is diagnostic (a request signed as `Bearer` // where the app only ever issues `Basic` is a bug worth seeing) and it is // not itself a secret. for scheme in AUTH_SCHEMES { if low[j..].starts_with(scheme.as_bytes()) { j = skip_spaces(low, j + scheme.len()); break; } } let start = j; let end = match quote { Some(q) => low[start..] .iter() .position(|b| *b == q) .map_or(low.len(), |n| start + n), None => value_end(low, start), }; // `password=` with nothing after it has nothing to hide, and replacing an // empty value would only make the line say less. (end > start).then_some((start, end)) } /// `Basic dXNlcjpwdw==` on its own, with no keyword in front of it — a header /// dumped by its value, or a `curl` line pasted into a message. fn bare_scheme_at(low: &[u8], i: usize) -> Option<(usize, usize)> { if i > 0 && is_word_byte(low[i - 1]) { return None; } let scheme = AUTH_SCHEMES .iter() .find(|scheme| low[i..].starts_with(scheme.as_bytes()))?; let start = skip_spaces(low, i + scheme.len()); let end = value_end(low, start); // Unlike the keyed rule, nothing here has established that a credential // was intended — and `basic` is an ordinary English word. "using basic // sRGB as the fallback" must survive, so the token itself has to look the // part. looks_like_a_credential(&low[start..end]).then_some((start, end)) } /// Whether an unkeyed token is credential-shaped. /// /// Base64 is long and drawn from a fixed alphabet; a word of prose is neither. /// Sixteen bytes is comfortably below the shortest thing that could be a real /// `Basic` credential — base64 of `a:b` is already 8, and a Nextcloud app /// password is 72 — and comfortably above the words that follow "basic" in a /// sentence. fn looks_like_a_credential(token: &[u8]) -> bool { token.len() >= 16 && token.iter().all(|b| { b.is_ascii_alphanumeric() || matches!(*b, b'+' | b'/' | b'=' | b'-' | b'_' | b'.') }) } /// The password in `https://duncan:hunter2@cloud.example/remote.php/dav`. /// /// URLs are logged constantly and this form survives every keyword rule, since /// the credential is punctuation-delimited rather than named. The host and the /// login are left in place: which server failed, and as whom, is the substance /// of a sync bug report. fn url_userinfo_at(low: &[u8], i: usize) -> Option<(usize, usize)> { if !low[i..].starts_with(b"://") { return None; } let authority = i + 3; // The authority ends at the path, the query, the fragment, or whatever // punctuation the surrounding prose used to end the URL. let end = low[authority..] .iter() .position(|b| matches!(*b, b'/' | b'?' | b'#') || is_value_delimiter(*b)) .map_or(low.len(), |n| authority + n); // The *last* `@`, because a userinfo may legally contain a percent-encoded // one and the host may not. let at = authority + low[authority..end].iter().rposition(|b| *b == b'@')?; let colon = authority + low[authority..at].iter().position(|b| *b == b':')?; (at > colon + 1).then_some((colon + 1, at)) } fn skip_spaces(low: &[u8], mut j: usize) -> usize { while matches!(low.get(j), Some(b' ' | b'\t')) { j += 1; } j } /// Where an unquoted value stops. fn value_end(low: &[u8], start: usize) -> usize { low[start..] .iter() .position(|b| is_value_delimiter(*b)) .map_or(low.len(), |n| start + n) } /// The punctuation that ends an unquoted value in every form this sees: JSON /// (`,` `}` `]` `"`), a query string (`&` `;`), a `Debug` impl (`,` `)` `}`), /// and prose (whitespace). fn is_value_delimiter(b: u8) -> bool { matches!( b, b' ' | b'\t' | b'\r' | b'\n' | b'"' | b'\'' | b',' | b';' | b'&' | b')' | b']' | b'}' | b'>' ) } /// What counts as "inside a word" for the boundary test. `_` is included so /// that `access_token` is one word rather than two, which is what stops the /// `token` rule from firing in the middle of it and redacting from there. fn is_word_byte(b: u8) -> bool { b.is_ascii_alphanumeric() || b == b'_' } #[cfg(test)] mod tests { use super::*; /// The strongest form of the assertion: not "the line changed" but "the /// secret is gone". A rule that redacted the wrong span would pass a /// weaker test. fn assert_gone(secret: &str, line: &str) -> String { let out = redact(line); assert!( !out.contains(secret), "redaction left the credential behind\n in: {line}\n out: {out}" ); assert!( out.contains(PLACEHOLDER), "nothing was marked as removed: {out}" ); out } #[test] fn the_credential_login_flow_returns_never_reaches_the_file() { // Verbatim the shape of `AppCredentials` as serde writes it — the one // credential this application actually holds (FR-NC-1). let out = assert_gone( "wYh3K8mLpQr2", r#"got {"server":"https://cloud.example","loginName":"duncan","appPassword":"wYh3K8mLpQr2"}"#, ); // The rest of the response is the diagnostic, and it survives. assert!(out.contains("cloud.example")); assert!(out.contains("duncan")); } #[test] fn a_debug_printed_credential_struct_is_scrubbed() { // The two-in-the-morning case: `log::debug!("{creds:?}")`. assert_gone( "wYh3K8mLpQr2", r#"AppCredentials { server: "https://cloud.example", login_name: "duncan", app_password: "wYh3K8mLpQr2" }"#, ); } #[test] fn a_basic_auth_header_keeps_its_scheme_and_loses_its_secret() { let out = assert_gone( "ZHVuY2FuOmh1bnRlcjI=", "PROPFIND /remote.php/dav authorization: Basic ZHVuY2FuOmh1bnRlcjI=", ); assert!( out.contains("Basic"), "which scheme was used is diagnostic, and is not the secret: {out}" ); } #[test] fn a_bearer_token_with_no_keyword_in_front_of_it_still_goes() { assert_gone( "eyJhbGciOiJIUzI1NiJ9", "sending Bearer eyJhbGciOiJIUzI1NiJ9", ); } #[test] fn a_password_in_a_url_goes_and_the_server_stays() { let out = assert_gone( "hunter2", "PUT https://duncan:hunter2@cloud.example/remote.php/dav/files/x.cr2 failed", ); assert!( out.contains("duncan") && out.contains("cloud.example"), "which server, and as whom, is the whole of a sync bug report: {out}" ); } #[test] fn a_query_string_token_goes() { assert_gone( "abc123def", "polling https://cloud.example/login/v2/poll?token=abc123def&x=1", ); } #[test] fn the_word_alone_is_not_a_secret() { // The over-redaction failure, which is how a scrubber makes a log // useless without ever being caught: these lines say nothing secret // and must come through untouched. for line in [ "password rejected by the server", "no token yet; still polling", "the tokenizer produced 41 tokens", "authorization failed with 401", "secretary@cloud.example added", // The unkeyed `Basic ` rule has no keyword to justify itself, and // `basic` is an ordinary word. This is the line it must not eat. "using basic sRGB as the fallback", "bearer of the news: 12 images", ] { assert_eq!(redact(line), line, "over-redacted: {line}"); } } #[test] fn a_photographs_path_is_not_a_secret() { // Stated as a test because it is a design decision and not an // oversight: NFR-SEC-2 lists credentials, and a log that cannot name // the file that failed to decode is not a diagnostic. See the module // documentation. let line = "decode failed: /home/duncan/Pictures/2026/Iceland/DSC_0431.NEF"; assert_eq!(redact(line), line); } #[test] fn several_secrets_in_one_line_all_go() { let out = redact(r#"{"password":"a1b2c3","token":"d4e5f6","user":"duncan"}"#); assert!(!out.contains("a1b2c3"), "{out}"); assert!(!out.contains("d4e5f6"), "{out}"); assert!(out.contains("duncan"), "{out}"); } #[test] fn redaction_keeps_the_line_shaped_like_what_it_replaced() { assert_eq!( redact(r#"{"password": "hunter2"}"#), r#"{"password": ""}"# ); assert_eq!(redact("password=hunter2"), "password="); } #[test] fn a_line_with_no_secret_survives_byte_for_byte() { // Including non-ASCII, which is the case the ASCII lowercasing exists // to protect: a `to_lowercase` here would shift byte offsets under a // Turkish dotted capital and slice the next character in half. for line in [ "opened /home/duncan/Bilder/Ölandsbron/DSC_0431.NEF", "İSTANBUL library scanned: 12 034 images", "", ] { assert_eq!(redact(line), line); } } #[test] fn a_url_without_a_credential_is_left_whole() { let line = "GET https://cloud.example/remote.php/dav/files/duncan/"; assert_eq!(redact(line), line); } #[test] fn an_email_address_is_not_a_url_userinfo() { let line = "sharing with duncan@tourolle.paris"; assert_eq!(redact(line), line); } }