Adds jray-project as a submodule at scripts/vendor/jray-project, so this repo runs the same extractor as every other component rather than its own copy, and gains the system spec that defines the PR/SR requirements its register traces up to. scripts/traceability-gate.sh is a thin wrapper holding only what is specific to this repo: UR/DR prefixes, .rs sources, and REPO_ROOT — which the shared gate cannot infer once vendored, since its default resolves to the submodule itself. Each override fails silently in a way that looks like "no work done" rather than "misconfigured", so the wrapper documents why each is needed. Annotates 35 units with TRACES tags, on the code that decides rather than every helper it calls. Coverage is 23/32 (71.9%) with no orphan tags. The nine untraced are genuinely unimplemented: UR-007 is plugin-side, UR-008 is federation, and UR-015..018 are the pending SR-003 schema bump. The gate caught a real error in the first pass: several tags separated IDs of different types with commas. A comma joins IDs within one type; a pipe separates types. Fixed, and the diagnostics are now clean. MIN_COVERAGE stays 0 deliberately. The gate still fails on orphan tags, a >100% ratio, a register parsing to nothing, or an empty source scan — raise the threshold as a ratchet once the remaining work lands. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
190 lines
6.6 KiB
Rust
190 lines
6.6 KiB
Rust
//! `GET /manifests/exists` and its batch form — UR-1.
|
|
//!
|
|
//! Deliberately a *separate, cheaper* endpoint from the fetch: it answers
|
|
//! "should I bother?" for a whole library sweep without transferring payloads,
|
|
//! and it is the endpoint a scheduled task will hammer. It is also the most
|
|
//! abuse-prone surface, since it doubles as an oracle for "does the community
|
|
//! have this title" — so it is rate-limited harder than the fetches and returns
|
|
//! no manifest content (§0).
|
|
|
|
use axum::extract::{Query, State};
|
|
use axum::http::HeaderMap;
|
|
use axum::response::{IntoResponse, Response};
|
|
use axum::Json;
|
|
use serde::{Deserialize, Serialize};
|
|
|
|
use super::LookupParams;
|
|
use crate::db::repo;
|
|
use crate::error::{ApiError, ApiResult};
|
|
use crate::matching::{self, StoredCut};
|
|
use crate::model::{IdentityType, MatchTier};
|
|
use crate::ratelimit::Surface;
|
|
use crate::state::{with_quota_headers, AppState};
|
|
|
|
/// §4: no manifest content, just availability and tier.
|
|
#[derive(Debug, Clone, Serialize)]
|
|
pub struct ExistsResponse {
|
|
pub exists: bool,
|
|
#[serde(skip_serializing_if = "Option::is_none")]
|
|
pub r#match: Option<&'static str>,
|
|
#[serde(skip_serializing_if = "Option::is_none")]
|
|
pub manifest_id: Option<String>,
|
|
#[serde(skip_serializing_if = "Option::is_none")]
|
|
pub actor_count: Option<i64>,
|
|
}
|
|
|
|
impl ExistsResponse {
|
|
fn absent() -> Self {
|
|
Self { exists: false, r#match: None, manifest_id: None, actor_count: None }
|
|
}
|
|
}
|
|
|
|
/// §4 batch form: up to 100 items.
|
|
///
|
|
/// Exists specifically so the §5 rate limit can be generous per *request* while
|
|
/// staying strict per *item*, and so a 2000-item library sweep is 20 requests
|
|
/// rather than 2000.
|
|
pub const MAX_BATCH_ITEMS: usize = 100;
|
|
|
|
#[derive(Debug, Deserialize)]
|
|
#[serde(deny_unknown_fields)]
|
|
pub struct BatchRequest {
|
|
pub items: Vec<LookupParams>,
|
|
}
|
|
|
|
#[derive(Debug, Serialize)]
|
|
pub struct BatchResponse {
|
|
/// Positional, matching the request order (§4).
|
|
pub results: Vec<ExistsResponse>,
|
|
}
|
|
|
|
/// TRACES: UR-001 | SR-001
|
|
pub async fn exists(
|
|
State(state): State<AppState>,
|
|
peer: crate::state::PeerIp,
|
|
headers: HeaderMap,
|
|
Query(params): Query<LookupParams>,
|
|
) -> ApiResult<Response> {
|
|
let ip = state.client_ip(&headers, peer.0);
|
|
let quota = state.check_limit(&ip, Surface::ExistsSingle)?;
|
|
let body = lookup_one(&state, ¶ms).await?;
|
|
Ok(with_quota_headers(Json(body).into_response(), quota))
|
|
}
|
|
|
|
/// TRACES: UR-001, UR-007 | SR-001 | PR-005
|
|
pub async fn exists_batch(
|
|
State(state): State<AppState>,
|
|
peer: crate::state::PeerIp,
|
|
headers: HeaderMap,
|
|
super::json::Json(req): super::json::Json<BatchRequest>,
|
|
) -> ApiResult<Response> {
|
|
if req.items.len() > MAX_BATCH_ITEMS {
|
|
return Err(ApiError::BadRequest(format!(
|
|
"items: at most {MAX_BATCH_ITEMS} per request, got {}",
|
|
req.items.len()
|
|
)));
|
|
}
|
|
if req.items.is_empty() {
|
|
return Err(ApiError::BadRequest("items: must not be empty".into()));
|
|
}
|
|
|
|
let ip = state.client_ip(&headers, peer.0);
|
|
let quota = state.check_limit(&ip, Surface::ExistsBatch)?;
|
|
|
|
let mut results = Vec::with_capacity(req.items.len());
|
|
for item in &req.items {
|
|
// A malformed item yields "absent" rather than failing the whole batch —
|
|
// a sweep of 100 items should not be lost to one bad entry.
|
|
results.push(lookup_one(&state, item).await.unwrap_or_else(|_| ExistsResponse::absent()));
|
|
}
|
|
|
|
Ok(with_quota_headers(Json(BatchResponse { results }).into_response(), quota))
|
|
}
|
|
|
|
async fn lookup_one(state: &AppState, params: &LookupParams) -> ApiResult<ExistsResponse> {
|
|
let Some((kind, tmdb_id, imdb_id)) = resolve_kind(params) else {
|
|
return Err(ApiError::BadRequest(
|
|
"requires tmdb_id/imdb_id, or series_tmdb_id with season and episode".into(),
|
|
));
|
|
};
|
|
|
|
let (season, episode) = match kind {
|
|
IdentityType::Movie => (None, None),
|
|
IdentityType::Episode => (params.season, params.episode),
|
|
};
|
|
let client_cut = params.client_cut();
|
|
|
|
let found = state
|
|
.db
|
|
.read(move |conn| {
|
|
let Some(title) = repo::find_title(conn, kind, tmdb_id.as_deref(), imdb_id.as_deref())?
|
|
else {
|
|
return Ok(None);
|
|
};
|
|
let candidates = repo::candidates_for_title(conn, &title.id, season, episode)?;
|
|
if candidates.is_empty() {
|
|
return Ok(None);
|
|
}
|
|
|
|
let cuts: Vec<(String, StoredCut)> = candidates
|
|
.iter()
|
|
.map(|m| {
|
|
(
|
|
m.id.clone(),
|
|
StoredCut { runtime_sec: m.runtime_sec, video_hash: m.video_hash.clone() },
|
|
)
|
|
})
|
|
.collect();
|
|
|
|
let Some((id, m)) = matching::best_match(&client_cut, &cuts) else {
|
|
return Ok(None);
|
|
};
|
|
let actor_count = repo::manifest_actor_ids(conn, &id)?.len() as i64;
|
|
Ok(Some((id, m.tier, actor_count)))
|
|
})
|
|
.await
|
|
.map_err(ApiError::Internal)?;
|
|
|
|
// §4: `exists: false` is returned with `200`, not `404` — absence is a normal
|
|
// answer to this question, and `404` would conflate "no manifest" with "bad
|
|
// route" for the client.
|
|
Ok(match found {
|
|
Some((id, tier, actor_count)) => ExistsResponse {
|
|
exists: true,
|
|
// With no cut parameters the answer is "some manifest exists" with
|
|
// `"match": "unknown"`; the client must still fetch to find out
|
|
// whether a cut aligns. This is the mode a library sweep uses (§4).
|
|
r#match: Some(tier.as_str()),
|
|
manifest_id: Some(id),
|
|
actor_count: Some(actor_count),
|
|
},
|
|
None => ExistsResponse::absent(),
|
|
})
|
|
}
|
|
|
|
/// Determines whether these parameters address a movie or an episode.
|
|
pub fn resolve_kind(
|
|
params: &LookupParams,
|
|
) -> Option<(IdentityType, Option<String>, Option<String>)> {
|
|
if params.series_tmdb_id.is_some() || params.series_imdb_id.is_some() {
|
|
// Episode coordinates are required alongside series identity; without
|
|
// them the caller wants the series bundle endpoint instead.
|
|
params.season?;
|
|
params.episode?;
|
|
return Some((
|
|
IdentityType::Episode,
|
|
params.series_tmdb_id.clone(),
|
|
params.series_imdb_id.clone(),
|
|
));
|
|
}
|
|
if params.tmdb_id.is_some() || params.imdb_id.is_some() {
|
|
return Some((IdentityType::Movie, params.tmdb_id.clone(), params.imdb_id.clone()));
|
|
}
|
|
None
|
|
}
|
|
|
|
/// Exposed for tests asserting the documented tier string.
|
|
pub fn tier_str(t: MatchTier) -> &'static str {
|
|
t.as_str()
|
|
}
|