Skip to main content

doiget_core/
http.rs

1// allow: outbound-network
2//! Centralized HTTP client wrapper. All `Source` impls fetch through here.
3//!
4//! Security defaults per `docs/SECURITY.md`:
5//!   - rustls TLS only (no openssl, no native-tls — enforced by `deny.toml`)
6//!   - HTTPS-only redirect policy (file://, data://, http:// rejected)
7//!   - Per-source redirect host allowlist (`docs/REDIRECT_ALLOWLIST.md`)
8//!   - Body size cap ([`crate::PDF_MAX_BYTES`] = 100 MB)
9//!   - Per-request timeouts (connect 10s, read 60s, total 300s)
10//!   - PDF magic-byte check on the first 5 bytes (`%PDF-`)
11//!   - User-Agent: `doiget/<version> (+https://github.com/QAtlasHub/doiget)`
12//!
13//! See `docs/SECURITY.md` §1.2-1.3 / §1.10 and `docs/REDIRECT_ALLOWLIST.md`.
14//!
15//! # Architectural note: per-source `reqwest::Client`
16//!
17//! `reqwest::redirect::Policy::custom` receives only an `Attempt` value, which
18//! exposes the next URL and previous URL chain but **not** the original
19//! request's headers. That makes the "tag the request with `X-Doiget-Source`
20//! and inspect it from inside the redirect closure" approach infeasible on
21//! `reqwest 0.13.x`. Instead, [`HttpClient`] holds one
22//! [`reqwest::Client`] per source — each client's redirect closure captures
23//! that source's [`SourceAllowlist`] so cross-source confusion is impossible
24//! by construction.
25
26use std::collections::HashMap;
27use std::sync::Arc;
28use std::sync::Once;
29use std::time::Duration;
30
31use bytes::{Bytes, BytesMut};
32use futures_util::StreamExt;
33use reqwest::redirect::Policy;
34use reqwest::{Client, ClientBuilder, Url};
35use thiserror::Error;
36
37use crate::{PDF_MAX_BYTES, VERSION};
38
39/// PDF magic-byte prefix per the PDF 1.7 specification (ISO 32000-1 §7.5.2).
40/// `b"%PDF-"`.
41const PDF_MAGIC: [u8; 5] = [0x25, 0x50, 0x44, 0x46, 0x2D];
42
43/// Hard cap on redirect chain length. Matches `reqwest`'s default of 10.
44/// Re-asserted here so the value is reviewed alongside the other security
45/// defaults in this module rather than inheriting silently from upstream.
46const MAX_REDIRECTS: usize = 10;
47
48/// Connect timeout per `docs/SECURITY.md` §1.2 (Slowloris row).
49const CONNECT_TIMEOUT: Duration = Duration::from_secs(10);
50
51/// Read (idle-between-bytes) timeout per `docs/SECURITY.md` §1.2.
52const READ_TIMEOUT: Duration = Duration::from_secs(60);
53
54/// Total per-request timeout per `docs/SECURITY.md` §1.2.
55const TOTAL_TIMEOUT: Duration = Duration::from_secs(300);
56
57/// Max retry attempts AFTER the first try, for transient failures only
58/// (connect/timeout/mid-stream network errors and the transient HTTP
59/// status set). 3 retries → up to 4 total attempts. See issue #117.
60const MAX_FETCH_RETRIES: u32 = 3;
61
62/// Base delay for the exponential backoff (`base * 2^attempt`, jittered).
63const RETRY_BASE_DELAY: Duration = Duration::from_millis(500);
64
65/// Hard ceiling on any single backoff / `Retry-After` sleep. Keeps the
66/// worst-case retry chain comfortably inside [`TOTAL_TIMEOUT`].
67const RETRY_MAX_DELAY: Duration = Duration::from_secs(30);
68
69/// HTTP status codes worth retrying: request timeout, rate-limited, and
70/// the transient 5xx family. A plain 500 is included because upstreams
71/// (Crossref/Unpaywall) intermittently 500 under load. 4xx other than
72/// 408/429 are caller/permanent and never retried.
73fn is_transient_status(code: u16) -> bool {
74    matches!(code, 408 | 429 | 500 | 502 | 503 | 504)
75}
76
77/// A `reqwest::Error` is transient iff it is a connect or timeout
78/// failure or a mid-body transfer error. Redirect-policy aborts
79/// (allowlist denial), builder errors, and decode errors are NOT
80/// transient — retrying them cannot help and would mask a real denial.
81fn reqwest_is_transient(e: &reqwest::Error) -> bool {
82    (e.is_timeout() || e.is_connect() || e.is_body()) && !e.is_redirect()
83}
84
85/// Parse a `Retry-After` header expressed as integer seconds (the
86/// HTTP-date form is accepted by the RFC but rare for these APIs and
87/// deliberately ignored for the MVP — we fall back to exponential
88/// backoff in that case). Capped at [`RETRY_MAX_DELAY`].
89fn parse_retry_after(headers: &reqwest::header::HeaderMap) -> Option<Duration> {
90    let secs: u64 = headers
91        .get(reqwest::header::RETRY_AFTER)?
92        .to_str()
93        .ok()?
94        .trim()
95        .parse()
96        .ok()?;
97    Some(Duration::from_secs(secs).min(RETRY_MAX_DELAY))
98}
99
100/// Exponential backoff with decorrelated jitter. `RETRY_BASE_DELAY *
101/// 2^attempt`, capped at [`RETRY_MAX_DELAY`], plus 0..base jitter so a
102/// fleet of clients does not thunder back in lockstep. Jitter is derived
103/// from the wall-clock subsec nanos rather than pulling in an RNG
104/// dependency — adequate decorrelation for backoff, not a security
105/// primitive.
106fn backoff_delay(attempt: u32) -> Duration {
107    let factor = 1u64 << attempt.min(20);
108    let base_ms = RETRY_BASE_DELAY.as_millis() as u64;
109    let capped_ms = base_ms
110        .saturating_mul(factor)
111        .min(RETRY_MAX_DELAY.as_millis() as u64);
112    let jitter_ms = std::time::SystemTime::now()
113        .duration_since(std::time::UNIX_EPOCH)
114        .map(|d| (d.subsec_nanos() as u64) % base_ms.max(1))
115        .unwrap_or(0);
116    Duration::from_millis(capped_ms.saturating_add(jitter_ms))
117}
118
119// ---------------------------------------------------------------------------
120// SourceAllowlist
121// ---------------------------------------------------------------------------
122
123/// Per-source allowlist entry. Matches the schema in
124/// What a [`HttpClient::probe`] observed (issue #407).
125///
126/// `body_bytes` is load-bearing, not decoration: a publisher WAF answers a
127/// scripted client with `202 Accepted` and an empty body, which a
128/// status-only report reads as success. Status plus body size separates
129/// "the publisher served me" from "the publisher is holding me at a bot
130/// challenge".
131#[derive(Debug, Clone, PartialEq, Eq)]
132pub struct ProbeOutcome {
133    /// HTTP status of the (post-redirect) response.
134    pub status: u16,
135    /// Bytes of body received.
136    pub body_bytes: usize,
137    /// Host of the final URL, after any allowlisted redirects.
138    pub final_host: Option<String>,
139}
140
141/// `docs/REDIRECT_ALLOWLIST.md` §2.
142#[derive(Debug, Clone)]
143#[non_exhaustive]
144pub struct SourceAllowlist {
145    /// Source key. MUST match a `source` value in `docs/SOURCES.md` §1
146    /// (e.g. `crossref`, `unpaywall`, `arxiv`).
147    pub source: String,
148    /// Each pattern is either a literal FQDN or a `*.<suffix>` glob (matches
149    /// the suffix and any subdomain — see `docs/REDIRECT_ALLOWLIST.md` §2.2
150    /// matching rule).
151    pub redirect_hosts: Vec<String>,
152}
153
154impl SourceAllowlist {
155    /// Construct a new allowlist entry.
156    pub fn new(source: impl Into<String>, redirect_hosts: Vec<String>) -> Self {
157        Self {
158            source: source.into(),
159            redirect_hosts,
160        }
161    }
162
163    /// Returns `true` if `host` matches any pattern in this allowlist.
164    ///
165    /// Matching is byte-level on the lowercased ASCII form of the host.
166    /// Callers MUST lowercase upstream; this method also lowercases as a
167    /// defense-in-depth measure but treats the result as ASCII (Punycode
168    /// is the caller's responsibility per `docs/REDIRECT_ALLOWLIST.md`
169    /// §2.2 rule 4).
170    pub fn matches(&self, host: &str) -> bool {
171        let host_lc = host.to_ascii_lowercase();
172        self.redirect_hosts
173            .iter()
174            .any(|pat| host_matches_pattern(&host_lc, pat))
175    }
176
177    /// Returns `true` if a request to `host` may proceed under this
178    /// allowlist: either it is on the list, or it is a transparent DOI
179    /// resolver ([`is_transparent_resolver`]).
180    ///
181    /// **This, not [`matches`](Self::matches), is what an adjudication site
182    /// calls.** `matches` answers "is this host on the list", which is a
183    /// question about the list; `permits` answers "may we go here", which is
184    /// the question every gate is actually asking. Keeping them separate
185    /// means the resolver set is not silently reported as part of any
186    /// source's `expected_hosts`.
187    ///
188    /// The five adjudication sites -- the pre-fetch OA-URL check in
189    /// `orchestrator`, the two redirect-policy closures, `probe`, and the
190    /// pre-check `doiget config doctor --network` runs before calling
191    /// `probe` -- all route through here so they cannot disagree. #533 was
192    /// found because only one of them was ever walked end to end, and the
193    /// doctor's was missed on the first pass at this very doc comment: it
194    /// said "four" while still calling `matches`, which would have had the
195    /// command that explains allowlist refusals reproduce #533 the moment a
196    /// resolver host entered its probe list.
197    #[must_use]
198    pub fn permits(&self, host: &str) -> bool {
199        is_transparent_resolver(host) || self.matches(host)
200    }
201}
202
203/// Hosts that are addressing, not hosting.
204///
205/// `doi.org` is the indirection layer every DOI passes through, and
206/// Unpaywall routinely reports it AS the OA location: for the gold cc-by
207/// paper in #533, `best_oa_location.url` is literally
208/// `https://doi.org/10.1002/pcn5.205` with no `url_for_pdf`. Adjudicating
209/// that hop as if it were a content host had two consequences, both wrong:
210///
211///   * the chain was refused at the FIRST hop, before it ever reached the
212///     publisher -- whose host, `*.wiley.com`, was already on the list; and
213///   * the denial's remediation told the user to allowlist `doi.org`, which
214///     does not widen the trusted surface toward one publisher. It removes
215///     the bound entirely, because every DOI in existence resolves through
216///     it. An agent following that advice would get the PDF and silently
217///     lose the invariant the allowlist exists to hold (ADR-0027).
218///
219/// These hosts are therefore FOLLOWED but never allowlisted, never named as
220/// remediation, and never counted as the source of the content. The host
221/// that actually serves the bytes is adjudicated exactly as before, so this
222/// is transparent to the invariant rather than an exception to it.
223///
224/// One edge the sentence above does not cover: a chain that TERMINATES at a
225/// resolver -- a `200` straight from `doi.org` rather than the `302` it
226/// exists to send -- is served by a host `permits` allowed and `matches`
227/// would not. The bound still holds in practice because these three hosts are
228/// operated by the DOI Foundation and CNRI and do not serve article bytes,
229/// which is why the set is closed and exact; but the invariant is "the
230/// resolver is trusted to redirect", not "only allowlisted hosts ever send
231/// bytes".
232///
233/// # Why a closed set and not "the terminal host"
234///
235/// #533's first suggestion was to adjudicate only the final host of a
236/// chain. That is a bigger change than it looks: it would let a chain
237/// traverse ANY host so long as it ended somewhere allowed, and every hop
238/// still sees the request. A named, closed set of resolvers keeps the bound.
239///
240/// # Why exact hosts and no wildcards
241///
242/// `*.doi.org` would sweep in `www.doi.org`, which is the DOI Foundation's
243/// website, not a resolver -- and any other subdomain the Foundation ever
244/// stands up. Each entry here is a host measured to 302 straight to the
245/// publisher (2026-08-30, `10.1002/pcn5.205`):
246///
247/// | host | what it is |
248/// |---|---|
249/// | `doi.org` | the canonical DOI resolver |
250/// | `dx.doi.org` | its long-standing alias, still in live metadata |
251/// | `hdl.handle.net` | the Handle System resolver `doi.org` proxies |
252const TRANSPARENT_RESOLVER_HOSTS: &[&str] = &["doi.org", "dx.doi.org", "hdl.handle.net"];
253
254/// Whether `host` is a DOI resolver rather than a content host (#533).
255///
256/// Exact match against `TRANSPARENT_RESOLVER_HOSTS`, deliberately without
257/// wildcard support: `evil-doi.org`, `doi.org.evil.test` and `www.doi.org`
258/// are all NOT resolvers, and the first two are what an attacker would
259/// register.
260#[must_use]
261pub fn is_transparent_resolver(host: &str) -> bool {
262    let host_lc = host.to_ascii_lowercase();
263    TRANSPARENT_RESOLVER_HOSTS.contains(&host_lc.as_str())
264}
265
266/// Returns `true` if `host` (already lowercased) matches `pattern` per
267/// `docs/REDIRECT_ALLOWLIST.md` §2.2.
268fn host_matches_pattern(host: &str, pattern: &str) -> bool {
269    let pat_lc = pattern.to_ascii_lowercase();
270    if let Some(suffix) = pat_lc.strip_prefix("*.") {
271        // Suffix-glob: matches `<suffix>` exactly OR `*.<suffix>`.
272        host == suffix || host.ends_with(&format!(".{}", suffix))
273    } else {
274        // Exact-FQDN: byte-identical (after lowercasing both sides).
275        host == pat_lc
276    }
277}
278
279/// Hard-coded Phase 1 allowlist for Tier 1 sources. Sourced from
280/// `docs/REDIRECT_ALLOWLIST.md` §3.
281///
282/// Marked `Phase 1; revisit during real fetches` in the spec — entries
283/// flagged `(unverified)` (e.g. arXiv subdomain redirect behavior) MUST be
284/// confirmed or removed before Phase 1 is closed; see §3.3 of the spec.
285pub fn tier_1_allowlist() -> Vec<SourceAllowlist> {
286    vec![
287        // §3.1 crossref
288        SourceAllowlist::new(
289            "crossref",
290            vec!["api.crossref.org".to_string(), "*.crossref.org".to_string()],
291        ),
292        // §3.2 unpaywall
293        SourceAllowlist::new("unpaywall", vec!["api.unpaywall.org".to_string()]),
294        // §3.3 arxiv
295        SourceAllowlist::new(
296            "arxiv",
297            vec![
298                "arxiv.org".to_string(),
299                "export.arxiv.org".to_string(),
300                "*.arxiv.org".to_string(),
301            ],
302        ),
303    ]
304}
305
306/// Always-compiled allowlist for **PubMed id resolution** (#500, ADR-0061):
307/// a PMID / PMCID is turned into its DOI by NCBI E-utilities. Not a fetch
308/// source either -- it answers "which DOI is this", never with content -- so
309/// it stays out of [`tier_1_allowlist`] and out of every fetch plan.
310pub fn pubmed_allowlist() -> Vec<SourceAllowlist> {
311    vec![SourceAllowlist::new(
312        crate::pubmed::NCBI,
313        vec!["eutils.ncbi.nlm.nih.gov".to_string()],
314    )]
315}
316
317/// Allowlist for the **preprint finders** that are not fetch sources. Each
318/// has its own gate, and each is asked only when the content leg found
319/// nothing; none of them serves content.
320///
321/// - bioRxiv / medRxiv `pubs` (#640), on `DOIGET_ENABLE_BIORXIV`: the
322///   preprint DOI behind a published DOI, then fetched through its own
323///   reported OA location, never from this host.
324/// - INSPIRE-HEP (#642), on `DOIGET_ENABLE_INSPIRE`: a record's arXiv id.
325/// - NASA ADS (#644), on a non-empty `DOIGET_ADS_TOKEN`: a record's arXiv id.
326pub fn preprint_allowlist() -> Vec<SourceAllowlist> {
327    vec![
328        SourceAllowlist::new("biorxiv", vec!["api.biorxiv.org".to_string()]),
329        // #642: INSPIRE-HEP's record for a DOI, read for its arXiv id only.
330        SourceAllowlist::new("inspire", vec!["inspirehep.net".to_string()]),
331        // #644: NASA ADS search, on the user's own token.
332        SourceAllowlist::new("ads", vec!["api.adsabs.harvard.edu".to_string()]),
333    ]
334}
335
336/// Always-compiled allowlist for **software citations** (#614, ADR-0058):
337/// `doiget cite` / `verify` on a GitHub repository or release URL. Kept out
338/// of [`tier_1_allowlist`] because it is not a fetch source -- a fetch
339/// plan's source list is read from that one, and GitHub must never appear
340/// in it. Only the CLI registers it; no MCP tool cites a URL.
341pub fn software_allowlist() -> Vec<SourceAllowlist> {
342    vec![
343        SourceAllowlist::new(
344            crate::software::GITHUB_API,
345            vec!["api.github.com".to_string()],
346        ),
347        SourceAllowlist::new(
348            crate::software::GITHUB_RAW,
349            vec!["raw.githubusercontent.com".to_string()],
350        ),
351    ]
352}
353
354/// Hard-coded Phase 4 allowlist for Tier 2 metadata sources (OpenAlex,
355/// Semantic Scholar, DOAJ). Sourced from `docs/SOURCES.md` §1 (the Tier 2
356/// table) and `docs/REDIRECT_ALLOWLIST.md` §3 (same redirect-allowlist
357/// policy as Tier 1, distinct source keys).
358///
359/// Returned hosts:
360///
361/// - `"openalex"` → `api.openalex.org` (production OpenAlex REST API).
362/// - `"semantic_scholar"` → `api.semanticscholar.org` (S2 Graph API base).
363/// - `"doaj"` → `doaj.org` + `*.doaj.org` (DOAJ public API; wildcard
364///   covers `api.doaj.org` and any v4+ subdomain split).
365///
366/// Per `docs/SOURCES.md` §4 "OpenAlex / Semantic Scholar / DOAJ", these
367/// sources are **metadata-only**: their `Source::fetch` impls MUST
368/// return `pdf_bytes: None`. The redirect closure in [`HttpClient`]
369/// uses this list to deny redirects to off-list hosts under each Tier
370/// 2 source key — identical mechanism to Tier 1, but the per-tool
371/// capability gate (`profile.metadata.openalex` etc.) is layered on
372/// top so the network surface remains capability-aware.
373pub fn tier_2_allowlist() -> Vec<SourceAllowlist> {
374    vec![
375        SourceAllowlist::new("openalex", vec!["api.openalex.org".to_string()]),
376        SourceAllowlist::new(
377            "semantic_scholar",
378            vec!["api.semanticscholar.org".to_string()],
379        ),
380        SourceAllowlist::new(
381            "doaj",
382            vec!["doaj.org".to_string(), "*.doaj.org".to_string()],
383        ),
384        // DataCite REST — DOI resolution for the second registration
385        // agency (#414). Distinct from `doaj.org`: that host serves
386        // article records, this one is the DOI registry API.
387        SourceAllowlist::new("datacite", vec!["api.datacite.org".to_string()]),
388        // HAL — French national OA repository, Solr-style search API
389        // (#418). `api.archives-ouvertes.fr` is the API host; the
390        // deposit landing pages live on `hal.science`, which is reached
391        // through the `oa-publisher` key (via `trust_oa_registries`),
392        // not this one.
393        SourceAllowlist::new("hal", vec!["api.archives-ouvertes.fr".to_string()]),
394        // OpenAIRE Graph API v1 (#416). The legacy `/search/publications`
395        // endpoint on the same host is unstable (503s) and deliberately
396        // unused; only the Graph path is called.
397        SourceAllowlist::new("openaire", vec!["api.openaire.eu".to_string()]),
398        // CORE REST v3 (#417). Optional bearer key; same host either way.
399        SourceAllowlist::new("core", vec!["api.core.ac.uk".to_string()]),
400        // Europe PMC REST (#415). This is the EBI API host; the OA PDF it
401        // points at lives on `europepmc.org`, which is already on the
402        // `oa-publisher` key and is where the download actually happens.
403        SourceAllowlist::new("europe-pmc", vec!["www.ebi.ac.uk".to_string()]),
404    ]
405}
406
407/// Always-compiled allowlist for the **discovery search** call path
408/// (ADR-0031).
409///
410/// Registers `api.openalex.org` under the `"openalex"` source key so the
411/// Tier-1 `discovery::paper_search` (`GET /works?search=`) can reach the
412/// endpoint in the **default `oa-only` binary** — unlike
413/// [`tier_2_allowlist`], which the CLI only wires in under
414/// `#[cfg(feature = "metadata")]` (#516; it was `citation` until then,
415/// which left every other Tier-2 source `UnknownSource` in a
416/// `metadata`-only build).
417///
418/// Discovery search is classified as Tier 1 OA metadata (read-only, never
419/// paywalled, never a PDF — same risk class as Crossref/Unpaywall), so its
420/// transport allowlist must exist regardless of the `metadata`/`citation`
421/// features (ADR-0031 D1/D2). The CLI's `build_http_client` extends the
422/// production allowlist with this **unconditionally**; in `metadata`
423/// builds [`tier_2_allowlist`] re-registers the identical
424/// `"openalex" → api.openalex.org` entry, which is a harmless idempotent
425/// `HashMap` overwrite in [`HttpClient::new`].
426pub fn discovery_allowlist() -> Vec<SourceAllowlist> {
427    vec![SourceAllowlist::new(
428        "openalex",
429        vec!["api.openalex.org".to_string()],
430    )]
431}
432
433/// Always-compiled allowlist for the **full-text extraction** call path
434/// (ADR-0032).
435///
436/// Registers `ar5iv.labs.arxiv.org` under a dedicated `"ar5iv"` source key
437/// so [`crate::paper_text::paper_text`] (`GET /html/<arxiv-id>`) can reach
438/// the ar5iv LaTeXML-XHTML renderer in the **default `oa-only` binary** —
439/// the same always-on posture as [`discovery_allowlist`].
440///
441/// The host is an arXiv subdomain (`*.arxiv.org` already matches it under
442/// the [`tier_1_allowlist`] `"arxiv"` key), so this adds no new
443/// registrable domain to the network surface — it only registers the host
444/// under a **distinct source key** so the provenance trail records that
445/// extracted text came from the ar5iv HTML renderer, not the arXiv
446/// PDF/Atom API (ADR-0032 D3). Full-text extraction is classified Tier-1
447/// OA metadata (read-only, OA, never a PDF reinterpretation), so its
448/// transport allowlist must exist regardless of any feature gate
449/// (ADR-0032 D2). The CLI's `build_http_client` extends the production
450/// allowlist with this **unconditionally**.
451pub fn fulltext_allowlist() -> Vec<SourceAllowlist> {
452    vec![SourceAllowlist::new(
453        "ar5iv",
454        vec!["ar5iv.labs.arxiv.org".to_string()],
455    )]
456}
457
458/// Hard-coded Phase 5a allowlist for the Springer Nature OA TDM
459/// source. Compile-gated by the `tdm-springer` Cargo feature so
460/// default release binaries never include the host pattern (per
461/// ADR-0002 and `docs/SOURCES.md` §3).
462///
463/// Returned entry:
464/// - `"tdm-springer"` → `api.springernature.com` (production base) +
465///   `*.springernature.com` (covers load-balancing subdomains; the
466///   redirect closure denies anything outside the wildcard).
467///
468/// Per `docs/SOURCES.md` §4 "TDM sources (Phase 5)", a fetch under
469/// this source key requires ALL THREE gates: Cargo feature compiled
470/// in, `DOIGET_KEY_SPRINGER` env var present, and
471/// `DOIGET_AGREE_TDM_SPRINGER=1`. The `CapabilityProfile` gate
472/// enforces the env-var pair; this allowlist is the transport gate.
473#[cfg(feature = "tdm-springer")]
474pub fn tier_3_springer_allowlist() -> Vec<SourceAllowlist> {
475    vec![SourceAllowlist::new(
476        "tdm-springer",
477        vec![
478            "api.springernature.com".to_string(),
479            "*.springernature.com".to_string(),
480        ],
481    )]
482}
483
484/// Hard-coded Phase 5b allowlist for the APS Harvest TDM source.
485/// Compile-gated by the `tdm-aps` Cargo feature so default release
486/// binaries never include the host pattern (per ADR-0002 and
487/// `docs/SOURCES.md` §3).
488///
489/// Returned entry:
490/// - `"tdm-aps"` → `harvest.aps.org` (production base) +
491///   `*.aps.org` (covers load-balancing subdomains; the redirect
492///   closure denies anything outside the wildcard).
493///
494/// Three-gate activation: Cargo feature compiled in,
495/// `DOIGET_KEY_APS` env var present, and `DOIGET_AGREE_TDM_APS=1`.
496/// The `CapabilityProfile` gate enforces the env-var pair; this
497/// allowlist is the transport gate.
498#[cfg(feature = "tdm-aps")]
499pub fn tier_3_aps_allowlist() -> Vec<SourceAllowlist> {
500    vec![SourceAllowlist::new(
501        "tdm-aps",
502        vec!["harvest.aps.org".to_string(), "*.aps.org".to_string()],
503    )]
504}
505
506/// Hard-coded Phase 5c allowlist for the Elsevier ScienceDirect TDM
507/// source. Compile-gated by the `tdm-elsevier` Cargo feature so
508/// default release binaries never include the host pattern (per
509/// ADR-0002 and `docs/SOURCES.md` §3).
510///
511/// Returned entry:
512/// - `"tdm-elsevier"` → `api.elsevier.com` (production base) +
513///   `*.elsevier.com` (covers load-balancing subdomains; the
514///   redirect closure denies anything outside the wildcard).
515///
516/// Three-gate activation: Cargo feature compiled in,
517/// `DOIGET_KEY_ELSEVIER` env var present, and
518/// `DOIGET_AGREE_TDM_ELSEVIER=1`. The `CapabilityProfile` gate
519/// enforces the env-var pair; this allowlist is the transport gate.
520#[cfg(feature = "tdm-elsevier")]
521pub fn tier_3_elsevier_allowlist() -> Vec<SourceAllowlist> {
522    vec![SourceAllowlist::new(
523        "tdm-elsevier",
524        vec!["api.elsevier.com".to_string(), "*.elsevier.com".to_string()],
525    )]
526}
527
528/// Hard-coded allowlist for the IEEE Xplore TDM source (#430).
529/// Compile-gated by the `tdm-ieee` Cargo feature so default release
530/// binaries never include the host pattern (per ADR-0002 and
531/// `docs/SOURCES.md` §3).
532///
533/// Returned entry:
534/// - `"tdm-ieee"` → `ieeexploreapi.ieee.org` (production base) +
535///   `*.ieee.org` (covers load-balancing subdomains; the redirect
536///   closure denies anything outside the wildcard).
537///
538/// Note the API host is deliberately NOT `ieeexplore.ieee.org`, the web
539/// front end: ADR-0039 records that the front end answers a scripted
540/// client with `202` and an empty body regardless of entitlement, which
541/// is why the TDM API is the supported route at all.
542///
543/// Three-gate activation: Cargo feature compiled in, `DOIGET_KEY_IEEE`
544/// env var present, and `DOIGET_AGREE_TDM_IEEE=1`. The
545/// `CapabilityProfile` gate enforces the env-var pair; this allowlist is
546/// the transport gate.
547#[cfg(feature = "tdm-ieee")]
548pub fn tier_3_ieee_allowlist() -> Vec<SourceAllowlist> {
549    vec![SourceAllowlist::new(
550        "tdm-ieee",
551        vec![
552            "ieeexploreapi.ieee.org".to_string(),
553            "*.ieee.org".to_string(),
554        ],
555    )]
556}
557
558/// Every Tier-3 TDM allowlist this build actually compiled in.
559///
560/// #454: the three per-publisher builders above had no caller. Both client
561/// builders — `doiget_cli::commands::fetch::build_http_client` and its MCP
562/// twin — assemble the client by naming a list of allowlist functions, and
563/// neither named these. So #444 taught the orchestrator to reach the
564/// sources and the transport then refused a source key it had never been
565/// told about: `UnknownSource { source_key: "tdm-aps" }`, which reads like
566/// an internal error rather than a missing registration.
567///
568/// One function rather than three `#[cfg]` blocks at each call site: a
569/// fourth publisher is then a single edit here, and the two client builders
570/// cannot drift apart — which is the drift that produced this bug.
571///
572/// Empty in a default build, where no Tier-3 feature is compiled in.
573#[must_use]
574pub fn tier_3_allowlists() -> Vec<SourceAllowlist> {
575    #[allow(unused_mut)]
576    let mut out: Vec<SourceAllowlist> = Vec::new();
577    #[cfg(feature = "tdm-aps")]
578    out.extend(tier_3_aps_allowlist());
579    #[cfg(feature = "tdm-elsevier")]
580    out.extend(tier_3_elsevier_allowlist());
581    #[cfg(feature = "tdm-springer")]
582    out.extend(tier_3_springer_allowlist());
583    #[cfg(feature = "tdm-ieee")]
584    out.extend(tier_3_ieee_allowlist());
585    out
586}
587
588/// Hard-coded Phase 1 allowlist for the synthetic `"oa-publisher"` source —
589/// the publisher / preprint / repository hosts to which Unpaywall's
590/// `best_oa_location.url` (or `url_for_pdf`) typically resolves.
591///
592/// **Status: informed-best-effort.** Per `docs/REDIRECT_ALLOWLIST.md` §3,
593/// every entry below is a documented OA-publisher host pulled from the
594/// public DOI / OA discovery surface as of this function's authoring; they
595/// are **not** a substitute for empirical validation. Entries marked
596/// `(unverified)` MUST be confirmed by a real fetch or removed before
597/// Phase 1 is closed.
598///
599/// The orchestrator (`doiget-cli::commands::fetch::fetch_doi`) calls
600/// [`HttpClient::fetch_pdf`] under the `"oa-publisher"` source key when
601/// Unpaywall returns an OA URL. If the OA host is not in this list, the
602/// PDF leg is denied (`HttpError::RedirectDenied`) and the orchestrator
603/// falls back to metadata-only success (the `informed-best-effort`
604/// posture from the spec section above).
605pub fn oa_publisher_allowlist() -> Vec<SourceAllowlist> {
606    vec![SourceAllowlist::new(
607        "oa-publisher",
608        vec![
609            // Springer Nature OA imprints. Springer / SpringerOpen / Nature
610            // OA URLs all resolve under one of these registrable suffixes.
611            // (unverified) — confirm by replaying real Unpaywall responses.
612            "*.springer.com".to_string(),
613            "*.springeropen.com".to_string(),
614            "*.springernature.com".to_string(),
615            "*.nature.com".to_string(),
616            // Wiley OA. (unverified)
617            "*.wiley.com".to_string(),
618            // Elsevier OA route only — the TDM gated path is a separate
619            // source (`tdm-elsevier`, Phase 5c) and is not covered here.
620            // (unverified)
621            "*.elsevier.com".to_string(),
622            "*.sciencedirect.com".to_string(),
623            // Frontiers. (unverified)
624            "*.frontiersin.org".to_string(),
625            // MDPI. (unverified)
626            "*.mdpi.com".to_string(),
627            // PLOS. (unverified)
628            "*.plos.org".to_string(),
629            // Preprint servers — biorxiv / medrxiv. (unverified)
630            "*.biorxiv.org".to_string(),
631            "*.medrxiv.org".to_string(),
632            // Europe PMC + NIH PMC. (unverified)
633            "europepmc.org".to_string(),
634            "*.europepmc.org".to_string(),
635            "*.nih.gov".to_string(),
636            "*.ncbi.nlm.nih.gov".to_string(),
637            // Physics-society / diamond-OA hosts. UNLIKE the entries
638            // above, these are EMPIRICALLY VERIFIED: a real `doiget batch`
639            // over 30 OpenAlex-OA finite-temperature-MPS DOIs observed
640            // Unpaywall `best_oa_location` resolving to these hosts and
641            // being denied (#193, REDIRECT_ALLOWLIST.md §3.4, ADR-0027).
642            // APS — journals.aps.org / link.aps.org (green & gold OA;
643            // society host; `*.aps.org` is also trusted under the separate
644            // `tdm-aps` Tier-3 source key WHEN that feature is compiled
645            // in — `tier_3_aps_allowlist` is `#[cfg(feature = "tdm-aps")]`
646            // and absent from default release builds).
647            "*.aps.org".to_string(),
648            // SciPost — diamond OA, community-run physics publisher.
649            "scipost.org".to_string(),
650            "*.scipost.org".to_string(),
651            // IOP Publishing — iopscience.iop.org (New J. Phys. etc.).
652            "*.iop.org".to_string(),
653            // DOAJ — the canonical redirect host for gold-OA journal
654            // content. ADR-0037: this domain was ALREADY trusted in this
655            // file under the `"doaj"` metadata key (`tier_2_allowlist`),
656            // which the CLI wires in only under
657            // `#[cfg(feature = "citation")]` — so the two keys disagreed
658            // about a host the project had already accepted, and a stock
659            // build could not reach it at all. Promoted here on the
660            // ADR-0027 precedent that made `*.aps.org` unconditional
661            // rather than feature-gated. The apex is listed separately
662            // because a single-suffix wildcard does not match it and the
663            // observed redirect (10.1109/access.2024.3495502, #405)
664            // targeted the bare apex.
665            "doaj.org".to_string(),
666            "*.doaj.org".to_string(),
667            // arXiv — already on the `arxiv` tier-1 allowlist, but the
668            // Unpaywall-driven path uses the `oa-publisher` source key,
669            // so we mirror the host list here too. See REDIRECT_ALLOWLIST.md
670            // §3.3 for the underlying entries.
671            "arxiv.org".to_string(),
672            "*.arxiv.org".to_string(),
673        ],
674    )]
675}
676
677// ---------------------------------------------------------------------------
678// HttpError
679// ---------------------------------------------------------------------------
680
681/// Errors that can arise during HTTP fetches.
682#[derive(Debug, Error)]
683#[non_exhaustive]
684pub enum HttpError {
685    /// Transport / DNS / TLS failure or other `reqwest`-level error. Note
686    /// that `reqwest` surfaces a redirect-policy abort (via `Attempt::error`)
687    /// as a `reqwest::Error` carrying the source error — callers seeing
688    /// `Network` for what they believed was a redirect violation should
689    /// inspect the inner error chain.
690    #[error("network error: {0}")]
691    Network(#[from] reqwest::Error),
692    /// Redirect target host did not match any pattern in the source's
693    /// `redirect_hosts`. See `docs/REDIRECT_ALLOWLIST.md` §2.2.
694    ///
695    /// Field naming: `source_key` rather than `source` because `thiserror`
696    /// auto-treats a field literally named `source` as a `#[source]` error
697    /// chain link (which would require the field to implement `std::error::Error`).
698    ///
699    /// `expected_hosts` carries a snapshot of the source's allowlist
700    /// patterns at the time of the denial — populated for the structured
701    /// `denial_context.expected` channel introduced by ADR-0023 §4
702    /// (NORMATIVE mapping table). Cloning the patterns into the error
703    /// keeps the `From<&HttpError> for Option<DenialContext>` impl from
704    /// having to re-look-up the allowlist by `source_key`. May be empty
705    /// when the rejection happened before any allowlist was matched
706    /// (e.g. URL had no host component at all).
707    #[error("redirect target {host} not in allowlist for source {source_key}")]
708    RedirectDenied {
709        /// Source key whose allowlist rejected the redirect.
710        source_key: String,
711        /// The lowercased host that was rejected.
712        host: String,
713        /// Snapshot of the source's `redirect_hosts` at denial time.
714        /// Surfaces as `denial_context.expected` (ADR-0023 §4).
715        expected_hosts: Vec<String>,
716    },
717    /// Redirect target had a scheme other than `https`. See
718    /// `docs/SECURITY.md` §1.3.
719    #[error("redirect to non-HTTPS scheme: {scheme}")]
720    InsecureRedirect {
721        /// The disallowed scheme (e.g. `http`, `file`, `data`).
722        scheme: String,
723    },
724    /// Body would exceed [`PDF_MAX_BYTES`] either by a `Content-Length`
725    /// hint or by accumulated streamed bytes. See `docs/SECURITY.md` §1.2.
726    #[error("body too large: {actual} bytes (cap = {cap})")]
727    OversizedBody {
728        /// Observed size (header value or accumulated bytes).
729        actual: u64,
730        /// Hard upper bound (always [`PDF_MAX_BYTES`]).
731        cap: u64,
732    },
733    /// PDF magic-byte mismatch — the body does not start with `%PDF-`.
734    /// We deliberately do NOT use `Content-Type` (publishers misbehave —
735    /// the magic byte is the trustworthy signal per `docs/SECURITY.md`
736    /// §1.2 "Magic-byte mismatch" row).
737    #[error("PDF magic-byte mismatch: got {got:?}")]
738    NotAPdf {
739        /// First five bytes of the response body (zero-padded if shorter).
740        got: [u8; 5],
741    },
742    /// Server returned a non-2xx status.
743    #[error("HTTP {status} from {url}")]
744    HttpStatus {
745        /// HTTP status code.
746        status: u16,
747        /// The URL that produced the status.
748        url: String,
749        /// The server's own `Retry-After`, in milliseconds, when it sent one
750        /// on the response that ended the attempt (#506).
751        ///
752        /// `parse_retry_after` already read this header, but only on the
753        /// retry path -- the terminal `return` discarded it, so by the time
754        /// the error reached a caller the number was gone and
755        /// `error.retry_after_ms` looked impossible to fill honestly. It is
756        /// not: the LAST response carries its own `Retry-After`, and that is
757        /// the one the caller should wait.
758        ///
759        /// `None` when the server sent no header. Deliberately not
760        /// substituted with `backoff_delay` -- doiget's internal backoff is a
761        /// guess about the server, and handing a caller a guess wearing the
762        /// name of a server-supplied value is the defect this field exists to
763        /// avoid.
764        retry_after_ms: Option<u64>,
765    },
766    /// No allowlist entry exists for this source. The caller asked
767    /// [`HttpClient`] to fetch on behalf of a source that wasn't passed to
768    /// [`HttpClient::new`].
769    ///
770    /// See note on `RedirectDenied` for why the field is `source_key`.
771    #[error("no allowlist registered for source {source_key}")]
772    UnknownSource {
773        /// The unregistered source key.
774        source_key: String,
775    },
776    /// A header name or value passed to
777    /// [`HttpClient::fetch_bytes_with_headers`] was not a valid HTTP
778    /// header. The header parser only accepts the visible-ASCII subset
779    /// per RFC 7230 §3.2; control characters and non-ASCII bytes are
780    /// rejected before the request is even built. Surfaces as
781    /// `ErrorCode::InternalError` at the public boundary (callers
782    /// supplying bad headers are responsible for fixing the call site;
783    /// not a denial in the ADR-0023 sense).
784    #[error("invalid HTTP header `{name}`: {reason}")]
785    InvalidHeader {
786        /// The header name as supplied by the caller.
787        name: String,
788        /// `"name"` or `"value"` — which side failed parsing.
789        reason: String,
790    },
791}
792
793// ---------------------------------------------------------------------------
794// HttpError -> Option<DenialContext>  (ADR-0023 §4 mapping table)
795// ---------------------------------------------------------------------------
796
797/// Map an [`HttpError`] reference to the structured [`crate::DenialContext`]
798/// channel introduced by ADR-0023.
799///
800/// Returns `Some(_)` for the four denial classes named in ADR-0023 §4
801/// (`RedirectDenied`, `OversizedBody`, `NotAPdf`, `InsecureRedirect`) and
802/// `None` for every other variant — `Network`, `HttpStatus`,
803/// `UnknownSource` are not denials in the ADR-0023 sense (they are
804/// transport / upstream / programming-error signals, not allowlist or
805/// cap rejections).
806///
807/// The `&HttpError` borrow form is used (rather than `HttpError`) so the
808/// caller — typically the orchestrator that already needs the original
809/// error for `error.message` and the `From<HttpError> for ErrorCode`
810/// collapse — does not have to clone the error to produce the optional
811/// structured side-channel.
812impl From<&HttpError> for Option<crate::DenialContext> {
813    fn from(e: &HttpError) -> Self {
814        use crate::{DenialContext, DenialReason};
815        match e {
816            HttpError::RedirectDenied {
817                source_key,
818                host,
819                expected_hosts,
820            } => Some(DenialContext {
821                reason: DenialReason::RedirectNotInAllowlist,
822                source: Some(source_key.clone()),
823                attempted: Some(host.clone()),
824                expected: Some(expected_hosts.clone()),
825                hop_index: None,
826                cap: None,
827                actual: None,
828            }),
829            HttpError::OversizedBody { actual, cap } => Some(DenialContext {
830                reason: DenialReason::SizeCapExceeded,
831                source: None,
832                attempted: None,
833                // The size-cap reason has no allowlist channel; use
834                // `None` to signal "field not populated by producer"
835                // rather than `Some(vec![])` (which would mean "explicit
836                // empty allowlist"). See `DenialContext::expected` docs.
837                expected: None,
838                hop_index: None,
839                cap: Some(*cap),
840                actual: Some(*actual),
841            }),
842            HttpError::NotAPdf { got } => Some(DenialContext {
843                reason: DenialReason::ContentTypeMismatch,
844                source: None,
845                // ADR-0023 §4 mapping table: hex-encode the first 5 bytes
846                // for the `attempted` field. `format!("{:02x}...")` is
847                // chosen over `hex::encode` to avoid pulling the
848                // additional dep into this conversion path; the result is
849                // bit-identical (lowercase, zero-padded).
850                attempted: Some(format!(
851                    "{:02x}{:02x}{:02x}{:02x}{:02x}",
852                    got[0], got[1], got[2], got[3], got[4]
853                )),
854                expected: Some(vec!["%PDF-".to_string()]),
855                hop_index: None,
856                cap: None,
857                actual: None,
858            }),
859            HttpError::InsecureRedirect { scheme } => Some(DenialContext {
860                reason: DenialReason::InsecureScheme,
861                source: None,
862                attempted: Some(format!("{}:...", scheme)),
863                expected: Some(vec!["https".to_string()]),
864                hop_index: None,
865                cap: None,
866                actual: None,
867            }),
868            // `reqwest` wraps a custom error returned by the redirect
869            // policy closure (`attempt.error(HttpError::RedirectDenied{..})`
870            // / `attempt.error(HttpError::InsecureRedirect{..})`) inside a
871            // `reqwest::Error`, which surfaces here as `HttpError::Network`.
872            // Without source-chain walking, production redirect denials —
873            // the most operationally important denial class — would never
874            // produce a `DenialContext`, defeating the whole point of
875            // ADR-0023.
876            //
877            // Walk the `std::error::Error::source()` chain on the inner
878            // `reqwest::Error` and downcast each link to `&HttpError`. If
879            // a wrapped `HttpError` is found, recurse via this same `From`
880            // impl. Otherwise the network error is a "real" transport /
881            // DNS / TLS failure with no denial semantics — return `None`.
882            //
883            // `std::error::Error::source(e)` is fully-qualified to
884            // disambiguate against the inherent (and unrelated)
885            // `reqwest::Error::source()`.
886            HttpError::Network(e) => {
887                let mut source: Option<&(dyn std::error::Error + 'static)> =
888                    std::error::Error::source(e);
889                while let Some(s) = source {
890                    if let Some(http_err) = s.downcast_ref::<HttpError>() {
891                        return Option::<crate::DenialContext>::from(http_err);
892                    }
893                    source = s.source();
894                }
895                None
896            }
897            // The remaining variants are not "denials" in the ADR-0023
898            // sense — HttpStatus/UnknownSource are upstream / programming-
899            // error signals; InvalidHeader is a caller-bug signal.
900            HttpError::HttpStatus { .. }
901            | HttpError::UnknownSource { .. }
902            | HttpError::InvalidHeader { .. } => None,
903        }
904    }
905}
906
907// ---------------------------------------------------------------------------
908// HttpClient
909// ---------------------------------------------------------------------------
910
911/// Workspace-wide HTTP client with the security defaults applied.
912///
913/// Internally holds one `reqwest::Client` per source. Construct via
914/// [`HttpClient::new`] with the full set of allowlists the calling process
915/// will need.
916#[derive(Clone, Debug)]
917pub struct HttpClient {
918    /// One [`reqwest::Client`] per source. Each client carries a redirect
919    /// policy that captures only that source's allowlist. `Arc` so cloning
920    /// is cheap.
921    clients: Arc<HashMap<String, Client>>,
922    /// The exact [`SourceAllowlist`] each per-source client was built from,
923    /// keyed by source. The redirect closure inside each `reqwest::Client`
924    /// captures its allowlist *by move*, so it cannot be read back from the
925    /// client itself. This map keeps the identical `SourceAllowlist`
926    /// available to callers that must perform a *pre-fetch* host check on a
927    /// metadata-discovered URL (issue #145 / `docs/REDIRECT_ALLOWLIST.md`
928    /// §1: the allowlist is consulted "on the OA URL discovered through
929    /// metadata sources before the actual PDF fetch is issued", not only on
930    /// redirect hops). Storing the same value here — rather than re-deriving
931    /// it from [`oa_publisher_allowlist`] at the call site — guarantees the
932    /// pre-check and the redirect closure can never drift, and that the
933    /// check works under the test constructors too (which register a
934    /// wiremock host as the allowlist).
935    allowlists: Arc<HashMap<String, SourceAllowlist>>,
936}
937
938impl HttpClient {
939    /// Build a client with rustls + redirect-allowlist + size cap +
940    /// timeouts.
941    ///
942    /// `allowlists` MUST cover every source whose URL might be passed in;
943    /// fetches against unregistered sources return
944    /// [`HttpError::UnknownSource`].
945    ///
946    /// # Errors
947    ///
948    /// Returns the underlying `reqwest::Error` if `ClientBuilder::build`
949    /// fails (typically a TLS-backend init failure).
950    pub fn new(allowlists: Vec<SourceAllowlist>) -> Result<Self, reqwest::Error> {
951        let ua = format!("doiget/{} (+https://github.com/QAtlasHub/doiget)", VERSION);
952        Self::new_with_user_agent(allowlists, &ua)
953    }
954
955    /// Build a client with a custom `User-Agent` header.
956    ///
957    /// Used by `doiget batch --user-agent` to override the default UA for
958    /// hosts that classify the default string as a bot.
959    pub fn new_with_user_agent(
960        allowlists: Vec<SourceAllowlist>,
961        user_agent: &str,
962    ) -> Result<Self, reqwest::Error> {
963        let mut clients = HashMap::with_capacity(allowlists.len());
964        let mut allowlist_map = HashMap::with_capacity(allowlists.len());
965        for entry in allowlists {
966            let source = entry.source.clone();
967            allowlist_map.insert(source.clone(), entry.clone());
968            let client = build_client(entry, user_agent)?;
969            clients.insert(source, client);
970        }
971        Ok(Self {
972            clients: Arc::new(clients),
973            allowlists: Arc::new(allowlist_map),
974        })
975    }
976
977    /// The [`SourceAllowlist`] this client was built with for `source`, or
978    /// `None` if `source` was not registered.
979    ///
980    /// This is the *identical* value captured by the per-source redirect
981    /// closure (see [`HttpClient`]'s `allowlists` field doc). It exists so
982    /// the orchestrator can apply the `docs/REDIRECT_ALLOWLIST.md` §1
983    /// pre-fetch host check on a metadata-discovered OA URL — the URL that
984    /// is fetched *without* necessarily passing through a redirect hop —
985    /// using the same source of truth the redirect closure uses, so the two
986    /// can never disagree. Callers MUST use this for the `"oa-publisher"`
987    /// leg only; the initial template-constructed URL is exempt per
988    /// `docs/REDIRECT_ALLOWLIST.md` §6.
989    pub fn source_allowlist(&self, source: &str) -> Option<&SourceAllowlist> {
990        self.allowlists.get(source)
991    }
992
993    /// Fetch a URL, treating it as a JSON or text body. Caps at
994    /// [`PDF_MAX_BYTES`].
995    ///
996    /// Returns the response body bytes plus the effective final URL after
997    /// redirects (post-allowlist verification — every hop has already been
998    /// validated by the time this returns).
999    ///
1000    /// # Errors
1001    ///
1002    /// Any [`HttpError`] variant.
1003    pub async fn fetch_bytes(&self, source: &str, url: Url) -> Result<(Bytes, Url), HttpError> {
1004        self.fetch_inner(source, url, &[], false).await
1005    }
1006
1007    /// Like [`Self::fetch_bytes`] but attaches additional request
1008    /// headers to the outgoing GET. The headers are validated up-front
1009    /// against the visible-ASCII subset (RFC 7230 §3.2); any failure
1010    /// returns [`HttpError::InvalidHeader`] before the request is sent.
1011    ///
1012    /// Used by Tier-3 TDM sources that authenticate via a header
1013    /// (APS Harvest `X-API-Key`, Elsevier ScienceDirect `X-ELS-APIKey`).
1014    /// Header values appear on the wire only — they are never logged.
1015    ///
1016    /// # Errors
1017    ///
1018    /// Any [`HttpError`] variant including [`HttpError::InvalidHeader`].
1019    pub async fn fetch_bytes_with_headers(
1020        &self,
1021        source: &str,
1022        url: Url,
1023        headers: &[(&str, &str)],
1024    ) -> Result<(Bytes, Url), HttpError> {
1025        self.fetch_inner(source, url, headers, false).await
1026    }
1027
1028    /// Fetch a URL expected to be a PDF. Same as [`Self::fetch_bytes`] plus
1029    /// the magic-byte check on the first 5 bytes
1030    /// (`%PDF-` = `[0x25, 0x50, 0x44, 0x46, 0x2D]`). Mismatch returns
1031    /// [`HttpError::NotAPdf`].
1032    ///
1033    /// # Errors
1034    ///
1035    /// Any [`HttpError`] variant including [`HttpError::NotAPdf`].
1036    pub async fn fetch_pdf(&self, source: &str, url: Url) -> Result<(Bytes, Url), HttpError> {
1037        self.fetch_inner(source, url, &[], true).await
1038    }
1039
1040    /// [`Self::fetch_pdf`] with request headers, for sources that
1041    /// authenticate by header rather than by query parameter.
1042    ///
1043    /// This is the pairing the Tier-3 content leg needs (#458): APS Harvest
1044    /// wants `X-API-Key` *and* `Accept: application/pdf` on the same request
1045    /// that must be magic-byte checked. `fetch_bytes_with_headers` would
1046    /// send the headers and skip the check, which is exactly how a
1047    /// publisher error page or a WAF holding response — both 200s with a
1048    /// body — would end up written to `<safekey>.pdf`.
1049    ///
1050    /// Header values are sent on the wire only; they are never logged and
1051    /// never echoed into an error message (#146).
1052    ///
1053    /// # Errors
1054    ///
1055    /// Any [`HttpError`] variant including [`HttpError::NotAPdf`].
1056    pub async fn fetch_pdf_with_headers(
1057        &self,
1058        source: &str,
1059        url: Url,
1060        headers: &[(&str, &str)],
1061    ) -> Result<(Bytes, Url), HttpError> {
1062        self.fetch_inner(source, url, headers, true).await
1063    }
1064
1065    /// Single diagnostic request against `url`, reporting what came back
1066    /// instead of turning it into an error.
1067    ///
1068    /// This is the primitive behind `doiget config doctor --network`
1069    /// (issue #407). It differs from [`Self::fetch_bytes`] in three ways
1070    /// that matter for a diagnostic:
1071    ///
1072    /// - **Non-2xx is data, not failure.** "403" is the answer to the
1073    ///   question the user is asking, so it comes back in
1074    ///   [`ProbeOutcome::status`] rather than as `HttpError::HttpStatus`.
1075    /// - **No retries.** A probe that silently retried would hide the
1076    ///   very flakiness it is meant to expose, and would multiply load on
1077    ///   a publisher for a question that is not a fetch.
1078    /// - **The host is checked against the source allowlist up front.**
1079    ///   A doctor that probed arbitrary user-supplied hosts would be an
1080    ///   SSRF gadget wearing a diagnostic hat; the allowlist is the same
1081    ///   one a real fetch would enforce, which is also what makes
1082    ///   "not allowlisted" a meaningful answer.
1083    ///
1084    /// The body is read so that [`ProbeOutcome::body_bytes`] can
1085    /// distinguish a real `200` from the `202` + empty-body holding
1086    /// response publisher WAFs return to scripted clients — the case in
1087    /// #407 that a status code alone cannot diagnose. The read is capped
1088    /// by the same size limits as any other fetch.
1089    ///
1090    /// # Errors
1091    ///
1092    /// [`HttpError::UnknownSource`] if `source` is not registered,
1093    /// [`HttpError::RedirectDenied`] if the host is off the allowlist
1094    /// (before any request is sent), or [`HttpError::Network`] for a
1095    /// transport failure — a timeout IS the diagnosis, so it is returned
1096    /// rather than retried.
1097    pub async fn probe(&self, source: &str, url: Url) -> Result<ProbeOutcome, HttpError> {
1098        let client = self
1099            .clients
1100            .get(source)
1101            .ok_or_else(|| HttpError::UnknownSource {
1102                source_key: source.to_string(),
1103            })?;
1104        let host = url.host_str().unwrap_or_default().to_string();
1105        let allow = self
1106            .source_allowlist(source)
1107            .ok_or_else(|| HttpError::UnknownSource {
1108                source_key: source.to_string(),
1109            })?;
1110        if !allow.permits(&host) {
1111            return Err(HttpError::RedirectDenied {
1112                source_key: source.to_string(),
1113                host,
1114                expected_hosts: allow.redirect_hosts.clone(),
1115            });
1116        }
1117        let response = client.get(url).send().await.map_err(HttpError::Network)?;
1118        let status = response.status().as_u16();
1119        let final_host = response.url().host_str().map(str::to_string);
1120        let body_bytes = response
1121            .bytes()
1122            .await
1123            .map(|b| b.len())
1124            .map_err(HttpError::Network)?;
1125        Ok(ProbeOutcome {
1126            status,
1127            body_bytes,
1128            final_host,
1129        })
1130    }
1131
1132    async fn fetch_inner(
1133        &self,
1134        source: &str,
1135        url: Url,
1136        headers: &[(&str, &str)],
1137        check_pdf_magic: bool,
1138    ) -> Result<(Bytes, Url), HttpError> {
1139        // Normalise legacy `http://` URLs returned by OpenAlex /
1140        // Unpaywall metadata before send. See `upgrade_http_to_https`
1141        // for the rationale (TLS posture preserved per ADR-0020) and
1142        // the loopback carve-out.
1143        let url = upgrade_http_to_https(url);
1144
1145        let client = self
1146            .clients
1147            .get(source)
1148            .ok_or_else(|| HttpError::UnknownSource {
1149                source_key: source.to_string(),
1150            })?;
1151
1152        // Parse headers up-front so an invalid name/value fails BEFORE
1153        // we touch the network. `HeaderName::from_bytes` / `HeaderValue::from_str`
1154        // accept the visible-ASCII subset only (RFC 7230 §3.2).
1155        let mut header_map = reqwest::header::HeaderMap::with_capacity(headers.len());
1156        for (name, value) in headers {
1157            let hn = reqwest::header::HeaderName::from_bytes(name.as_bytes()).map_err(|_| {
1158                HttpError::InvalidHeader {
1159                    name: (*name).to_string(),
1160                    reason: "name".to_string(),
1161                }
1162            })?;
1163            let hv = reqwest::header::HeaderValue::from_str(value).map_err(|_| {
1164                HttpError::InvalidHeader {
1165                    name: (*name).to_string(),
1166                    reason: "value".to_string(),
1167                }
1168            })?;
1169            header_map.insert(hn, hv);
1170        }
1171
1172        // Bounded retry loop (issue #117). Only transient classes are
1173        // retried — connect/timeout/mid-stream network errors and the
1174        // transient HTTP status set. Allowlist denials, NotAPdf,
1175        // OversizedBody, 4xx (non-408/429) are deterministic and return
1176        // on the first occurrence. GET is idempotent so a retried
1177        // attempt re-streams the body from scratch.
1178        let mut attempt: u32 = 0;
1179        loop {
1180            let send_result = client
1181                .get(url.clone())
1182                .headers(header_map.clone())
1183                .send()
1184                .await;
1185            let response = match send_result {
1186                Ok(r) => r,
1187                Err(e) => {
1188                    if attempt < MAX_FETCH_RETRIES && reqwest_is_transient(&e) {
1189                        let d = backoff_delay(attempt);
1190                        tracing::warn!(
1191                            source,
1192                            attempt,
1193                            delay_ms = d.as_millis() as u64,
1194                            error = %e,
1195                            "transient send failure; retrying"
1196                        );
1197                        tokio::time::sleep(d).await;
1198                        attempt += 1;
1199                        continue;
1200                    }
1201                    return Err(HttpError::Network(e));
1202                }
1203            };
1204            let final_url = response.url().clone();
1205
1206            // Status check before body read so we can fail fast.
1207            let status = response.status();
1208            if !status.is_success() {
1209                let code = status.as_u16();
1210                if attempt < MAX_FETCH_RETRIES && is_transient_status(code) {
1211                    // Prefer the server's `Retry-After` over our backoff
1212                    // when present (429/503 commonly carry it).
1213                    let d = parse_retry_after(response.headers())
1214                        .unwrap_or_else(|| backoff_delay(attempt));
1215                    tracing::warn!(
1216                        source,
1217                        attempt,
1218                        status = code,
1219                        delay_ms = d.as_millis() as u64,
1220                        "transient HTTP status; retrying"
1221                    );
1222                    tokio::time::sleep(d).await;
1223                    attempt += 1;
1224                    continue;
1225                }
1226                return Err(HttpError::HttpStatus {
1227                    status: code,
1228                    // Read from the response that ENDED the attempt, so it is
1229                    // the server's number rather than ours (#506).
1230                    retry_after_ms: parse_retry_after(response.headers())
1231                        .map(|d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)),
1232                    // Issue #146: Springer Nature authenticates via an
1233                    // `api_key` URL query parameter, and IEEE via
1234                    // `apikey` (#430) — neither documents a header
1235                    // path. This error string is logged and may
1236                    // surface to the user, so strip either spelling
1237                    // before it leaves the client. A no-op for every
1238                    // other source, none of which puts a secret in
1239                    // the query string.
1240                    url: redact_api_key_query(&final_url),
1241                });
1242            }
1243
1244            // Content-Length fast-path: if header is present and exceeds
1245            // the cap, fail without reading any body (deterministic — not
1246            // retried). Per `docs/SECURITY.md` §1.2.
1247            if let Some(len) = response.content_length() {
1248                if len > PDF_MAX_BYTES {
1249                    return Err(HttpError::OversizedBody {
1250                        actual: len,
1251                        cap: PDF_MAX_BYTES,
1252                    });
1253                }
1254            }
1255
1256            // Stream body and enforce the cap as bytes accumulate. A
1257            // mid-stream transport error is transient (retry); an
1258            // oversized body is deterministic (return).
1259            let mut buf = BytesMut::new();
1260            let mut stream = response.bytes_stream();
1261            let mut oversized_at: Option<u64> = None;
1262            let mut stream_err: Option<reqwest::Error> = None;
1263            while let Some(chunk) = stream.next().await {
1264                let chunk = match chunk {
1265                    Ok(c) => c,
1266                    Err(e) => {
1267                        stream_err = Some(e);
1268                        break;
1269                    }
1270                };
1271                let projected = (buf.len() as u64).saturating_add(chunk.len() as u64);
1272                if projected > PDF_MAX_BYTES {
1273                    oversized_at = Some(projected);
1274                    break;
1275                }
1276                buf.extend_from_slice(&chunk);
1277            }
1278            if let Some(actual) = oversized_at {
1279                return Err(HttpError::OversizedBody {
1280                    actual,
1281                    cap: PDF_MAX_BYTES,
1282                });
1283            }
1284            if let Some(e) = stream_err {
1285                if attempt < MAX_FETCH_RETRIES && reqwest_is_transient(&e) {
1286                    let d = backoff_delay(attempt);
1287                    tracing::warn!(
1288                        source,
1289                        attempt,
1290                        delay_ms = d.as_millis() as u64,
1291                        error = %e,
1292                        "transient mid-stream failure; retrying"
1293                    );
1294                    tokio::time::sleep(d).await;
1295                    attempt += 1;
1296                    continue;
1297                }
1298                return Err(HttpError::Network(e));
1299            }
1300            let body = buf.freeze();
1301
1302            if check_pdf_magic {
1303                let mut got = [0u8; 5];
1304                let n = body.len().min(5);
1305                got[..n].copy_from_slice(&body[..n]);
1306                if got != PDF_MAGIC {
1307                    return Err(HttpError::NotAPdf { got });
1308                }
1309            }
1310
1311            return Ok((body, final_url));
1312        }
1313    }
1314}
1315
1316/// Return `url` rendered as a string with the value of any `api_key`
1317/// query parameter replaced by `REDACTED` (issue #146).
1318///
1319/// Springer Nature's TDM API authenticates **only** via an `api_key`
1320/// query parameter — there is no header-auth path upstream — so the key
1321/// is unavoidably in the request URL. This keeps it out of *our* log
1322/// and error sinks (the `HttpError::HttpStatus` string in particular,
1323/// which is `tracing`-logged and can surface to the user). It is a
1324/// structural no-op for every other source, none of which carry a
1325/// secret in the query string. Other pairs and their order are
1326/// preserved; a URL with no `api_key` pair is rendered unchanged.
1327fn redact_api_key_query(url: &url::Url) -> String {
1328    /// Every spelling a source puts a secret under. Springer uses
1329    /// `api_key`; IEEE (#430) uses `apikey`, one word. A source that
1330    /// invents a third spelling and does not add it here leaks its key
1331    /// into `HttpError::HttpStatus`, which is `tracing`-logged.
1332    const API_KEY_PARAMS: &[&str] = &["api_key", "apikey"];
1333    let is_secret = |k: &str| API_KEY_PARAMS.contains(&k);
1334    if url.query_pairs().all(|(k, _)| !is_secret(&k)) {
1335        return url.to_string();
1336    }
1337    let mut redacted = url.clone();
1338    let pairs: Vec<(String, String)> = url
1339        .query_pairs()
1340        .map(|(k, v)| {
1341            if is_secret(&k) {
1342                (k.into_owned(), "REDACTED".to_string())
1343            } else {
1344                (k.into_owned(), v.into_owned())
1345            }
1346        })
1347        .collect();
1348    redacted.query_pairs_mut().clear().extend_pairs(pairs);
1349    redacted.to_string()
1350}
1351
1352/// Test-oriented [`HttpClient`] constructor. Originally `cfg(test)`; now
1353/// also reachable from the `doiget-cli` orchestrator's integration tests
1354/// (which live outside this crate and therefore cannot see `cfg(test)`-gated
1355/// items). The constructor name retains its `for_tests_allow_http` signal —
1356/// production code MUST use [`HttpClient::new`] with [`tier_1_allowlist`].
1357#[allow(clippy::expect_used)]
1358impl HttpClient {
1359    /// Build a test-oriented `HttpClient` against an `http://` wiremock
1360    /// origin. The redirect closure still rejects insecure schemes — we only
1361    /// relax `https_only` at the connection level so wiremock can serve.
1362    /// This is acceptable because the redirect closure (which is the
1363    /// security-load-bearing path) is exercised by the
1364    /// `redirect_to_http_is_rejected_by_closure` test below.
1365    ///
1366    /// Production callers MUST use [`HttpClient::new`] with
1367    /// [`tier_1_allowlist`] — the `for_tests_allow_http` suffix is the load-
1368    /// bearing signal that this constructor lifts the initial-leg HTTPS-only
1369    /// requirement.
1370    pub fn new_for_tests_allow_http(source: &str, allowlist_host: &str) -> Self {
1371        let allowlist = SourceAllowlist::new(source, vec![allowlist_host.to_string()]);
1372        let client = build_client_allow_http(allowlist.clone()).expect("test client builds");
1373        let mut map = HashMap::new();
1374        let mut allowlist_map = HashMap::new();
1375        allowlist_map.insert(allowlist.source.clone(), allowlist.clone());
1376        map.insert(allowlist.source.clone(), client);
1377        Self {
1378            clients: Arc::new(map),
1379            allowlists: Arc::new(allowlist_map),
1380        }
1381    }
1382
1383    /// Multi-source variant of [`HttpClient::new_for_tests_allow_http`].
1384    ///
1385    /// Builds a relaxed-`https_only` client per `(source, allowlist_host)`
1386    /// pair. Used by the `doiget-cli` orchestrator's integration tests when
1387    /// more than one upstream needs to be wiremocked simultaneously
1388    /// (e.g. Crossref + Unpaywall against two different mock servers).
1389    /// Production callers MUST use [`HttpClient::new`] with
1390    /// [`tier_1_allowlist`].
1391    pub fn new_for_tests_allow_http_multi(entries: &[(&str, &str)]) -> Self {
1392        let mut map = HashMap::with_capacity(entries.len());
1393        let mut allowlist_map = HashMap::with_capacity(entries.len());
1394        for (source, host) in entries {
1395            let allowlist = SourceAllowlist::new(*source, vec![host.to_string()]);
1396            let client = build_client_allow_http(allowlist.clone()).expect("test client builds");
1397            allowlist_map.insert(allowlist.source.clone(), allowlist.clone());
1398            map.insert(allowlist.source.clone(), client);
1399        }
1400        Self {
1401            clients: Arc::new(map),
1402            allowlists: Arc::new(allowlist_map),
1403        }
1404    }
1405}
1406
1407fn build_client_allow_http(allowlist: SourceAllowlist) -> Result<Client, reqwest::Error> {
1408    ensure_crypto_provider();
1409    let allowlist_for_closure = allowlist.clone();
1410    let redirect_policy = Policy::custom(move |attempt| {
1411        let scheme = attempt.url().scheme().to_string();
1412        let host_opt = attempt.url().host_str().map(|h| h.to_ascii_lowercase());
1413        let prev_count = attempt.previous().len();
1414        if scheme != "https" {
1415            return attempt.error(HttpError::InsecureRedirect { scheme });
1416        }
1417        if prev_count >= MAX_REDIRECTS {
1418            return attempt.stop();
1419        }
1420        let host = match host_opt {
1421            Some(h) => h,
1422            None => {
1423                return attempt.error(HttpError::RedirectDenied {
1424                    source_key: allowlist_for_closure.source.clone(),
1425                    host: String::new(),
1426                    expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
1427                });
1428            }
1429        };
1430        if !allowlist_for_closure.permits(&host) {
1431            return attempt.error(HttpError::RedirectDenied {
1432                source_key: allowlist_for_closure.source.clone(),
1433                host,
1434                expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
1435            });
1436        }
1437        attempt.follow()
1438    });
1439    ClientBuilder::new()
1440        // `https_only(false)` only at this scope — production builders
1441        // (the public `HttpClient::new`) keep it on.
1442        .https_only(false)
1443        .redirect(redirect_policy)
1444        .connect_timeout(CONNECT_TIMEOUT)
1445        .timeout(TOTAL_TIMEOUT)
1446        .read_timeout(READ_TIMEOUT)
1447        .user_agent(format!(
1448            "doiget/{} (+https://github.com/QAtlasHub/doiget)",
1449            VERSION
1450        ))
1451        .tls_backend_rustls()
1452        .build()
1453}
1454
1455// ---------------------------------------------------------------------------
1456// ClientBuilder helpers
1457// ---------------------------------------------------------------------------
1458
1459/// Install the `ring` `rustls` crypto provider as the process default,
1460/// exactly once.
1461///
1462/// reqwest is built with the `rustls-no-provider` feature (ADR-0020
1463/// Amendment 1: drop aws-lc-rs so `cargo install` needs no cmake/C
1464/// toolchain and musl-static builds cleanly). With no bundled provider,
1465/// `reqwest::ClientBuilder::build` calls
1466/// `rustls::crypto::CryptoProvider::get_default()` and **panics**
1467/// (`"No provider set"`) unless a process-default provider was installed
1468/// first. Every client constructor below calls this; the `Once` makes it
1469/// safe to invoke from many sites and from concurrent tests.
1470fn ensure_crypto_provider() {
1471    static INIT: Once = Once::new();
1472    INIT.call_once(|| {
1473        // `install_default` errors only if a provider is already set;
1474        // under `Once` that is unreachable, but ignore it rather than
1475        // panic (another linked crate could have installed one first).
1476        let _ = rustls::crypto::ring::default_provider().install_default();
1477    });
1478}
1479
1480/// Public entry point for callers that build their own `reqwest::Client`
1481/// outside of [`HttpClient`] and need the process-default TLS provider
1482/// installed first (ADR-0020 Amendment 1).
1483///
1484/// Safe to call multiple times; the underlying `Once` makes it idempotent.
1485pub fn init_tls() {
1486    ensure_crypto_provider();
1487}
1488
1489/// Upgrade an `http://` URL to `https://` for legacy publisher
1490/// metadata. Loopback hosts (`localhost`, any RFC 6761 `.localhost`
1491/// TLD subdomain, `127.0.0.0/8`, `::1`, IPv4-mapped IPv6 loopback)
1492/// are returned unchanged so the `new_for_tests_allow_http*` wiremock
1493/// path continues to talk plain HTTP to the local fixture server.
1494///
1495/// Non-`http` schemes (`https`, `file`, anything else) and cannot-be-
1496/// base URLs are returned unchanged. The function is total: it never
1497/// panics and never returns an error.
1498///
1499/// # Audit / posture
1500///
1501/// On a successful upgrade the function emits a `tracing::info!` event
1502/// so the rewrite appears in the operator's default-level structured
1503/// log. On the (in-practice unreachable) `set_scheme` failure path a
1504/// `tracing::warn!` event is emitted before returning the original
1505/// URL; the production client's `https_only(true)` then rejects the
1506/// send with a clear network error, preserving the TLS posture
1507/// established by ADR-0020.
1508///
1509/// # `Domain("localhost")` arm subtlety
1510///
1511/// The url crate resolves the bare host `localhost` to `127.0.0.1`
1512/// (Ipv4 variant) when parsing an `http://` URL, so the `Domain` arm
1513/// does NOT fire for that case (the `Ipv4` arm catches it). The arm
1514/// IS load-bearing for the RFC 6761 `.localhost` TLD (e.g.
1515/// `myservice.localhost`, `api.localhost`), which the url crate does
1516/// NOT auto-resolve to an IP and keeps as `Host::Domain`.
1517fn upgrade_http_to_https(url: Url) -> Url {
1518    if url.scheme() != "http" {
1519        return url;
1520    }
1521    match url.host() {
1522        None => {
1523            // Cannot-be-base URL (e.g. `http:foo`) — `set_scheme`
1524            // would reject the conversion.
1525            return url;
1526        }
1527        Some(url::Host::Domain(d)) if is_localhost_domain(d) => return url,
1528        Some(url::Host::Ipv4(ip)) if ip.is_loopback() => return url,
1529        Some(url::Host::Ipv6(ip)) if is_ipv6_loopback(ip) => return url,
1530        Some(_) => {}
1531    }
1532    let mut upgraded = url.clone();
1533    if upgraded.set_scheme("https").is_err() {
1534        // url-crate `set_scheme` is documented to fail only for
1535        // cannot-be-base URLs and a few cross-family transitions;
1536        // `http -> https` is supported because both are "special"
1537        // schemes. The fallback below is defence-in-depth.
1538        tracing::warn!(
1539            url = %url,
1540            "set_scheme(http -> https) failed unexpectedly; \
1541             sending original URL — https_only(true) will reject",
1542        );
1543        return url;
1544    }
1545    tracing::info!(
1546        original = %url,
1547        upgraded = %upgraded,
1548        "upgraded http -> https for legacy publisher metadata"
1549    );
1550    upgraded
1551}
1552
1553/// `true` for the `localhost` literal and any RFC 6761 `.localhost`
1554/// TLD subdomain (`myservice.localhost`, `api.localhost`, etc.).
1555/// ASCII-case-insensitive per host-name conventions.
1556fn is_localhost_domain(d: &str) -> bool {
1557    if d.eq_ignore_ascii_case("localhost") {
1558        return true;
1559    }
1560    let suffix = ".localhost";
1561    let d_bytes = d.as_bytes();
1562    let s_bytes = suffix.as_bytes();
1563    if d_bytes.len() <= s_bytes.len() {
1564        return false;
1565    }
1566    let tail = &d_bytes[d_bytes.len() - s_bytes.len()..];
1567    tail.eq_ignore_ascii_case(s_bytes)
1568}
1569
1570/// `true` for `::1` and any IPv4-mapped loopback
1571/// (`::ffff:127.0.0.0/8`). `Ipv6Addr::is_loopback()` covers only `::1`,
1572/// so dual-stack callers that hit `[::ffff:127.0.0.1]` would otherwise
1573/// be silently upgraded.
1574fn is_ipv6_loopback(ip: std::net::Ipv6Addr) -> bool {
1575    if ip.is_loopback() {
1576        return true;
1577    }
1578    matches!(ip.to_ipv4_mapped(), Some(v4) if v4.is_loopback())
1579}
1580
1581fn build_client(allowlist: SourceAllowlist, ua: &str) -> Result<Client, reqwest::Error> {
1582    ensure_crypto_provider();
1583
1584    let user_agent = ua.to_string();
1585
1586    // Redirect policy: capture the per-source allowlist by value. The
1587    // closure is called for every redirect hop — there is no global
1588    // fallback, every hop is checked. Hard cap at MAX_REDIRECTS via the
1589    // attempt counter (mirrors reqwest's built-in limit).
1590    let allowlist_for_closure = allowlist.clone();
1591    let redirect_policy = Policy::custom(move |attempt| {
1592        // Inspect the candidate URL via owned copies so we can move
1593        // `attempt` into `error()` / `follow()` / `stop()` later without
1594        // the borrow checker complaining about an outstanding borrow of
1595        // `attempt`.
1596        let scheme = attempt.url().scheme().to_string();
1597        let host_opt = attempt.url().host_str().map(|h| h.to_ascii_lowercase());
1598        let prev_count = attempt.previous().len();
1599
1600        // 1. Reject non-HTTPS up front. The `https_only(true)` builder
1601        //    flag below also catches this, but we want the dedicated
1602        //    `InsecureRedirect` error path (not a generic `https_only`
1603        //    abort) — see `docs/SECURITY.md` §1.3.
1604        if scheme != "https" {
1605            return attempt.error(HttpError::InsecureRedirect { scheme });
1606        }
1607
1608        // 2. Hop limit (`docs/SECURITY.md` §1.3 redirect_limit row).
1609        if prev_count >= MAX_REDIRECTS {
1610            return attempt.stop();
1611        }
1612
1613        // 3. Allowlist check on the candidate target host.
1614        //    `host_str()` is `None` for URLs without a host (e.g. data
1615        //    URIs); treat that as an allowlist miss.
1616        let host = match host_opt {
1617            Some(h) => h,
1618            None => {
1619                return attempt.error(HttpError::RedirectDenied {
1620                    source_key: allowlist_for_closure.source.clone(),
1621                    host: String::new(),
1622                    expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
1623                });
1624            }
1625        };
1626        if !allowlist_for_closure.permits(&host) {
1627            return attempt.error(HttpError::RedirectDenied {
1628                source_key: allowlist_for_closure.source.clone(),
1629                host,
1630                expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
1631            });
1632        }
1633
1634        attempt.follow()
1635    });
1636
1637    ClientBuilder::new()
1638        .https_only(true)
1639        .redirect(redirect_policy)
1640        .connect_timeout(CONNECT_TIMEOUT)
1641        .timeout(TOTAL_TIMEOUT)
1642        .read_timeout(READ_TIMEOUT)
1643        .user_agent(user_agent)
1644        // `tls_backend_rustls()` is the non-deprecated equivalent of the
1645        // older `use_rustls_tls()`. The workspace pins reqwest with
1646        // `rustls-no-provider` (ADR-0020 Amendment 1), so this is a
1647        // re-assertion at builder level rather than a feature switch; the
1648        // `ring` provider installed by `ensure_crypto_provider()` above
1649        // is what reqwest picks up via `CryptoProvider::get_default()`.
1650        .tls_backend_rustls()
1651        .build()
1652}
1653
1654// ---------------------------------------------------------------------------
1655// Tests
1656// ---------------------------------------------------------------------------
1657
1658#[cfg(test)]
1659#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
1660mod tests {
1661    use super::*;
1662
1663    /// Every Tier-2 `Source` MUST have a transport allowlist entry under
1664    /// its own `name()`, or `HttpClient::fetch_bytes` rejects it with
1665    /// `UnknownSource` in production.
1666    ///
1667    /// This was not hypothetical: #414 shipped `DataCiteSource` with no
1668    /// `tier_2_allowlist` entry. Every unit test passed because they build
1669    /// their client with `new_for_tests_allow_http("datacite", ..)`, which
1670    /// registers the key itself — so the tests could not see the gap, and
1671    /// only a real fetch would have. Enumerating the names here means
1672    /// adding a source without its allowlist entry fails at `cargo test`.
1673    /// #442 sibling of the Tier-2 guard. A source with no allowlist entry
1674    /// fails `UnknownSource` in production while every unit test passes,
1675    /// because `new_for_tests_allow_http` registers the key itself — the
1676    /// DataCite near-miss in 0.8.8. Now that Tier 3 is actually reached,
1677    /// the same trap applies to it.
1678    ///
1679    /// **This guard is necessary and not sufficient**, and #454 is the
1680    /// proof: it asserts the builder *returns* the key, which stayed true
1681    /// for three releases while no client was ever handed the list, so a
1682    /// production fetch died at exactly the `UnknownSource` described
1683    /// above. The sufficient half is
1684    /// `the_production_client_registers_every_tier_3_source_key`, in
1685    /// `doiget-cli` and `doiget-mcp` — it asserts the client, which is the
1686    /// object the fetch goes through. Keep both: this one localises the
1687    /// failure to the list, that one catches the list never arriving.
1688    #[cfg(any(
1689        feature = "tdm-aps",
1690        feature = "tdm-elsevier",
1691        feature = "tdm-springer",
1692        feature = "tdm-ieee"
1693    ))]
1694    #[test]
1695    fn every_tier_3_source_has_a_transport_allowlist_entry() {
1696        use crate::source::Source as _;
1697        let mut checked = 0_usize;
1698
1699        #[cfg(feature = "tdm-aps")]
1700        {
1701            let src = crate::sources::tdm_aps::TdmApsSource::new();
1702            let reg: Vec<String> = tier_3_aps_allowlist()
1703                .iter()
1704                .map(|a| a.source.clone())
1705                .collect();
1706            assert!(
1707                reg.iter().any(|r| r == src.name()),
1708                "source `{}` has no tier_3_aps_allowlist entry; a production fetch would fail \
1709                    UnknownSource. registered: {reg:?}",
1710                src.name()
1711            );
1712            checked += 1;
1713        }
1714        #[cfg(feature = "tdm-elsevier")]
1715        {
1716            let src = crate::sources::tdm_elsevier::TdmElsevierSource::new();
1717            let reg: Vec<String> = tier_3_elsevier_allowlist()
1718                .iter()
1719                .map(|a| a.source.clone())
1720                .collect();
1721            assert!(
1722                reg.iter().any(|r| r == src.name()),
1723                "source `{}` has no tier_3_elsevier_allowlist entry; a production fetch would \
1724                    fail UnknownSource. registered: {reg:?}",
1725                src.name()
1726            );
1727            checked += 1;
1728        }
1729        #[cfg(feature = "tdm-springer")]
1730        {
1731            let src = crate::sources::tdm_springer::TdmSpringerSource::new();
1732            let reg: Vec<String> = tier_3_springer_allowlist()
1733                .iter()
1734                .map(|a| a.source.clone())
1735                .collect();
1736            assert!(
1737                reg.iter().any(|r| r == src.name()),
1738                "source `{}` has no tier_3_springer_allowlist entry; a production fetch would \
1739                    fail UnknownSource. registered: {reg:?}",
1740                src.name()
1741            );
1742            checked += 1;
1743        }
1744        #[cfg(feature = "tdm-ieee")]
1745        {
1746            let src = crate::sources::tdm_ieee::TdmIeeeSource::new();
1747            let reg: Vec<String> = tier_3_ieee_allowlist()
1748                .iter()
1749                .map(|a| a.source.clone())
1750                .collect();
1751            assert!(
1752                reg.iter().any(|r| r == src.name()),
1753                "source `{}` has no tier_3_ieee_allowlist entry; a production fetch would \
1754                    fail UnknownSource. registered: {reg:?}",
1755                src.name()
1756            );
1757            checked += 1;
1758        }
1759
1760        assert!(checked > 0, "the guard must have checked something");
1761    }
1762
1763    #[test]
1764    #[cfg(feature = "metadata")]
1765    fn every_tier_2_source_has_a_transport_allowlist_entry() {
1766        use crate::source::Source as _;
1767        // Bind first: `name()` borrows from the source, so the values must
1768        // outlive the collection.
1769        let openalex = crate::sources::openalex::OpenalexSource::new(String::new());
1770        let s2 = crate::sources::s2::S2Source::new(None);
1771        let doaj = crate::sources::doaj::DoajSource::new();
1772        let datacite = crate::sources::datacite::DataCiteSource::new();
1773        let hal = crate::sources::hal::HalSource::new();
1774        let openaire = crate::sources::openaire::OpenAireSource::new();
1775        let core = crate::sources::core_oa::CoreSource::new();
1776        let epmc = crate::sources::europepmc::EuropePmcSource::new();
1777        let names: Vec<&str> = vec![
1778            openalex.name(),
1779            s2.name(),
1780            doaj.name(),
1781            datacite.name(),
1782            hal.name(),
1783            openaire.name(),
1784            core.name(),
1785            epmc.name(),
1786        ];
1787        let registered: Vec<String> = tier_2_allowlist()
1788            .iter()
1789            .map(|a| a.source.clone())
1790            .collect();
1791        for n in names {
1792            assert!(
1793                registered.iter().any(|r| r == n),
1794                "source `{n}` has no tier_2_allowlist entry; a production fetch would fail \
1795                    UnknownSource. registered: {registered:?}"
1796            );
1797        }
1798    }
1799
1800    /// ADR-0037: `doaj.org` must be reachable on the `oa-publisher` key with
1801    /// NO config file and NO feature flags — that is the whole point of
1802    /// promoting it. Pinned on the apex specifically: a single-suffix
1803    /// wildcard does not match an apex, and the redirect that motivated
1804    /// #405 (10.1109/access.2024.3495502, IEEE Access gold OA) targeted the
1805    /// bare apex.
1806    /// #533, reproduced against the REAL allowlist rather than a fixture.
1807    ///
1808    /// Harada & Kato 2024 is gold OA, cc-by, and Unpaywall's
1809    /// `best_oa_location.url` for it is literally `https://doi.org/10.1002/
1810    /// pcn5.205` with no `url_for_pdf` (verified against the live API,
1811    /// 2026-08-30). The chain was refused at that first hop -- while
1812    /// `*.wiley.com`, where it lands, was already on the list.
1813    #[test]
1814    fn a_doi_resolver_hop_is_permitted_without_being_allowlisted() {
1815        let lists = oa_publisher_allowlist();
1816        let oa = lists
1817            .iter()
1818            .find(|a| a.source == "oa-publisher")
1819            .expect("oa-publisher entry");
1820
1821        // The two questions stay distinct: a resolver is permitted, but it is
1822        // NOT on the list and must never be reported as if it were.
1823        assert!(oa.permits("doi.org"), "the hop the report was refused at");
1824        assert!(
1825            !oa.matches("doi.org"),
1826            "a resolver must not be ON the allowlist -- `expected_hosts` is \
1827             shown to the user as what to trust, and doi.org is every DOI"
1828        );
1829        assert!(
1830            !oa.redirect_hosts.iter().any(|h| h.contains("doi.org")),
1831            "and it must not leak into the expected-host list: {:?}",
1832            oa.redirect_hosts
1833        );
1834
1835        // The host the chain actually lands on was trusted all along. This is
1836        // what makes the refusal a layer error rather than a missing entry.
1837        assert!(
1838            oa.permits("onlinelibrary.wiley.com"),
1839            "the publisher was already covered by *.wiley.com: {:?}",
1840            oa.redirect_hosts
1841        );
1842
1843        // And the gate still gates.
1844        assert!(!oa.permits("evil.example.com"));
1845    }
1846
1847    /// Exact hosts, no wildcards. `*.doi.org` would sweep in `www.doi.org`,
1848    /// which is the DOI Foundation's website and not a resolver, plus
1849    /// whatever else is ever stood up there; and the impostor hosts below are
1850    /// precisely what an attacker registers.
1851    #[test]
1852    fn resolver_impostor_hosts_are_not_transparent() {
1853        for host in [
1854            "doi.org",
1855            "dx.doi.org",
1856            "hdl.handle.net",
1857            "DOI.ORG",
1858            "Dx.Doi.Org",
1859        ] {
1860            assert!(is_transparent_resolver(host), "{host} is a resolver");
1861        }
1862        for host in [
1863            "www.doi.org",
1864            "doi.org.evil.test",
1865            "evil-doi.org",
1866            "notdoi.org",
1867            "a.doi.org",
1868            "handle.net",
1869            "",
1870        ] {
1871            assert!(!is_transparent_resolver(host), "{host} is NOT a resolver");
1872        }
1873    }
1874
1875    /// Transparency is a property of the resolver, not of one source: a
1876    /// resolver hop is addressing wherever it appears.
1877    #[test]
1878    fn every_tier_1_source_treats_a_resolver_hop_as_addressing() {
1879        for a in tier_1_allowlist() {
1880            assert!(
1881                a.permits("doi.org"),
1882                "source {} refuses the addressing layer",
1883                a.source
1884            );
1885            assert!(
1886                !a.matches("doi.org"),
1887                "source {} lists a resolver as a content host",
1888                a.source
1889            );
1890        }
1891    }
1892
1893    #[test]
1894    fn doaj_is_on_the_default_oa_publisher_allowlist() {
1895        let lists = oa_publisher_allowlist();
1896        let oa = lists
1897            .iter()
1898            .find(|a| a.source == "oa-publisher")
1899            .expect("oa-publisher entry");
1900        assert!(
1901            oa.matches("doaj.org"),
1902            "apex must match: {:?}",
1903            oa.redirect_hosts
1904        );
1905        assert!(oa.matches("www.doaj.org"), "subdomains must match");
1906        assert!(
1907            !oa.matches("doaj.org.evil.test"),
1908            "suffix confusion must not match"
1909        );
1910    }
1911
1912    /// The `"doaj"` metadata key and the `"oa-publisher"` PDF-redirect key
1913    /// must now agree about DOAJ. Their disagreement was the defect ADR-0037
1914    /// fixed; this pins that they cannot silently drift apart again.
1915    #[test]
1916    fn doaj_metadata_and_oa_publisher_keys_agree() {
1917        let meta = tier_2_allowlist();
1918        let doaj = meta.iter().find(|a| a.source == "doaj").expect("doaj key");
1919        let lists = oa_publisher_allowlist();
1920        let oa = lists
1921            .iter()
1922            .find(|a| a.source == "oa-publisher")
1923            .expect("oa key");
1924        for pat in &doaj.redirect_hosts {
1925            let sample = pat.strip_prefix("*.").unwrap_or(pat);
1926            assert!(
1927                oa.matches(sample),
1928                "{pat} is trusted on the doaj key but not on oa-publisher"
1929            );
1930        }
1931    }
1932    use wiremock::matchers::{method, path};
1933    use wiremock::{Mock, MockServer, ResponseTemplate};
1934
1935    // ---------------------------------------------------------------
1936    // http -> https scheme upgrade (#220) — pure unit tests, no network.
1937    // ---------------------------------------------------------------
1938
1939    #[test]
1940    fn upgrade_http_to_https_rewrites_public_http_url() {
1941        let input = Url::parse("http://link.aps.org/pdf/10.1103/PhysRev.123.456").unwrap();
1942        let out = upgrade_http_to_https(input.clone());
1943        assert_eq!(out.scheme(), "https");
1944        assert_eq!(out.host_str(), Some("link.aps.org"));
1945        assert_eq!(out.path(), "/pdf/10.1103/PhysRev.123.456");
1946    }
1947
1948    #[test]
1949    fn upgrade_http_to_https_preserves_port_path_query_fragment() {
1950        let input = Url::parse("http://example.org:8080/a/b?q=1#frag").unwrap();
1951        let out = upgrade_http_to_https(input);
1952        assert_eq!(out.as_str(), "https://example.org:8080/a/b?q=1#frag");
1953    }
1954
1955    #[test]
1956    fn upgrade_http_to_https_is_idempotent_on_https() {
1957        let input = Url::parse("https://api.crossref.org/works/10.1234/foo").unwrap();
1958        let out = upgrade_http_to_https(input.clone());
1959        assert_eq!(out, input);
1960    }
1961
1962    #[test]
1963    fn upgrade_http_to_https_skips_localhost() {
1964        // wiremock binds to `127.0.0.1:PORT`; the loopback exception
1965        // is the load-bearing rule that keeps `new_for_tests_allow_http*`
1966        // working alongside the production fetch path.
1967        let input = Url::parse("http://localhost:7878/pdf").unwrap();
1968        let out = upgrade_http_to_https(input.clone());
1969        assert_eq!(out, input, "localhost MUST NOT be upgraded");
1970    }
1971
1972    #[test]
1973    fn upgrade_http_to_https_skips_127_loopback_block() {
1974        for host in ["127.0.0.1", "127.0.0.42", "127.255.255.254"] {
1975            let raw = format!("http://{host}:1234/x");
1976            let input = Url::parse(&raw).unwrap();
1977            let out = upgrade_http_to_https(input.clone());
1978            assert_eq!(out, input, "host `{host}` MUST NOT be upgraded");
1979        }
1980    }
1981
1982    #[test]
1983    fn upgrade_http_to_https_skips_ipv6_loopback() {
1984        let input = Url::parse("http://[::1]:9000/path").unwrap();
1985        let out = upgrade_http_to_https(input.clone());
1986        assert_eq!(out, input, "IPv6 loopback MUST NOT be upgraded");
1987    }
1988
1989    #[test]
1990    fn upgrade_http_to_https_preserves_case_in_path() {
1991        // Some publishers (e.g. APS legacy redirects) use mixed-case
1992        // path segments; upgrade must NOT lowercase or canonicalise.
1993        let input = Url::parse("http://link.aps.org/PDF/10.1103/PhysRevB.109.045136").unwrap();
1994        let out = upgrade_http_to_https(input);
1995        assert_eq!(out.path(), "/PDF/10.1103/PhysRevB.109.045136");
1996    }
1997
1998    // ---- Review-pass extensions ------------------------------------
1999
2000    #[test]
2001    fn upgrade_http_to_https_skips_dot_localhost_tld() {
2002        // RFC 6761 reserves the entire `.localhost` TLD for loopback.
2003        // A developer running `http://myservice.localhost:8080/` MUST
2004        // NOT see their URL silently upgraded to https.
2005        for raw in [
2006            "http://myservice.localhost/",
2007            "http://api.localhost:8080/x",
2008            "http://a.b.LOCALHOST/y",
2009        ] {
2010            let input = Url::parse(raw).unwrap();
2011            let out = upgrade_http_to_https(input.clone());
2012            assert_eq!(out, input, "{raw} MUST NOT be upgraded");
2013        }
2014    }
2015
2016    #[test]
2017    fn upgrade_http_to_https_skips_ipv4_mapped_ipv6_loopback() {
2018        // `::ffff:127.0.0.1` is the IPv4-mapped IPv6 form of 127.0.0.1.
2019        // `Ipv6Addr::is_loopback()` alone returns false for this form,
2020        // so dual-stack callers binding wiremock to it would be
2021        // silently upgraded without the `to_ipv4_mapped()` check.
2022        for raw in [
2023            "http://[::ffff:127.0.0.1]:9000/x",
2024            "http://[::ffff:127.0.0.42]/y",
2025        ] {
2026            let input = Url::parse(raw).unwrap();
2027            let out = upgrade_http_to_https(input.clone());
2028            assert_eq!(out, input, "{raw} MUST NOT be upgraded");
2029        }
2030    }
2031
2032    #[test]
2033    fn upgrade_http_to_https_is_noop_on_non_http_schemes() {
2034        // The first guard (`url.scheme() != "http"`) covers everything
2035        // that isn't http: https (idempotent), file, data, ftp...
2036        for raw in [
2037            "https://api.crossref.org/works/10.1234/foo",
2038            "file:///etc/passwd",
2039            "data:text/plain,hello",
2040            "ftp://ftp.example.org/papers/",
2041        ] {
2042            let input = Url::parse(raw).unwrap();
2043            let out = upgrade_http_to_https(input.clone());
2044            assert_eq!(
2045                out, input,
2046                "{raw} non-http scheme MUST be returned unchanged"
2047            );
2048        }
2049    }
2050
2051    #[test]
2052    fn upgrade_http_to_https_http_url_always_has_host() {
2053        // The url crate's parser enforces authority for "special"
2054        // schemes (`http`, `https`, `ws`, `wss`, `ftp`, `file`).
2055        // `Url::parse("http:foo")` synthesises a Domain("foo")
2056        // authority, so an http URL with `host() == None` is
2057        // unreachable from `Url::parse`. The `None` arm in
2058        // `upgrade_http_to_https` is defence-in-depth only — pinned
2059        // here so a future url-crate behavior change is caught.
2060        let url = Url::parse("http:foo").expect("parse");
2061        assert!(
2062            url.host().is_some(),
2063            "http URLs always carry a host per WHATWG URL spec"
2064        );
2065        // The fn still produces a sensible result (upgrade applies).
2066        let out = upgrade_http_to_https(url.clone());
2067        assert_eq!(out.scheme(), "https");
2068    }
2069
2070    #[test]
2071    fn upgrade_http_to_https_skips_localhost_case_insensitive() {
2072        // The literal `localhost` is resolved by the url crate to
2073        // `127.0.0.1` (Ipv4) at parse time for `http://` URLs, so the
2074        // Ipv4 arm catches lowercase. The Domain-arm coverage is
2075        // load-bearing only for the `.localhost` TLD case, but we
2076        // still pin the casefold semantics in case the url crate
2077        // changes its parsing rules.
2078        for raw in ["http://LOCALHOST/", "http://Localhost:8080/x"] {
2079            let input = Url::parse(raw).unwrap();
2080            let out = upgrade_http_to_https(input.clone());
2081            assert_eq!(out, input, "{raw} MUST NOT be upgraded");
2082        }
2083    }
2084
2085    #[test]
2086    fn is_localhost_domain_matches_literal_and_tld_suffix() {
2087        assert!(is_localhost_domain("localhost"));
2088        assert!(is_localhost_domain("LOCALHOST"));
2089        assert!(is_localhost_domain("api.localhost"));
2090        assert!(is_localhost_domain("nested.api.localhost"));
2091        assert!(is_localhost_domain("X.LocalHost"));
2092        assert!(!is_localhost_domain("localhost.example.org"));
2093        assert!(!is_localhost_domain("notlocalhost"));
2094        assert!(!is_localhost_domain(""));
2095        assert!(!is_localhost_domain(".localhost")); // empty label not valid
2096    }
2097
2098    #[test]
2099    fn is_ipv6_loopback_covers_both_pure_and_mapped() {
2100        use std::net::Ipv6Addr;
2101        assert!(is_ipv6_loopback(Ipv6Addr::LOCALHOST)); // ::1
2102        assert!(is_ipv6_loopback("::ffff:127.0.0.1".parse().unwrap()));
2103        assert!(is_ipv6_loopback("::ffff:127.0.0.42".parse().unwrap()));
2104        assert!(!is_ipv6_loopback("::".parse().unwrap()));
2105        assert!(!is_ipv6_loopback("2001:db8::1".parse().unwrap()));
2106        // IPv4-mapped non-loopback must NOT be considered loopback.
2107        assert!(!is_ipv6_loopback("::ffff:1.2.3.4".parse().unwrap()));
2108    }
2109
2110    // ---------------------------------------------------------------
2111    // Allowlist matching — pure unit tests, no network.
2112    // ---------------------------------------------------------------
2113
2114    #[test]
2115    fn tier_1_allowlist_includes_crossref() {
2116        let lists = tier_1_allowlist();
2117        let crossref = lists
2118            .iter()
2119            .find(|a| a.source == "crossref")
2120            .expect("crossref entry");
2121        assert!(
2122            crossref
2123                .redirect_hosts
2124                .iter()
2125                .any(|h| h.contains("crossref.org")),
2126            "crossref allowlist must contain a crossref.org pattern; got {:?}",
2127            crossref.redirect_hosts,
2128        );
2129    }
2130
2131    #[test]
2132    fn tier_1_allowlist_includes_unpaywall_and_arxiv() {
2133        let lists = tier_1_allowlist();
2134        assert!(lists.iter().any(|a| a.source == "unpaywall"));
2135        assert!(lists.iter().any(|a| a.source == "arxiv"));
2136    }
2137
2138    #[test]
2139    fn fulltext_allowlist_registers_ar5iv_host_under_distinct_key() {
2140        // ADR-0032 D3: the ar5iv renderer is registered under its own
2141        // `"ar5iv"` source key (not `"arxiv"`) so provenance distinguishes
2142        // full-text HTML from the arXiv PDF/Atom API.
2143        let lists = fulltext_allowlist();
2144        assert_eq!(lists.len(), 1, "exactly one full-text source entry");
2145        let ar5iv = &lists[0];
2146        assert_eq!(ar5iv.source, "ar5iv");
2147        assert!(ar5iv.matches("ar5iv.labs.arxiv.org"));
2148        // It is also an arXiv subdomain — the existing `*.arxiv.org` glob
2149        // already covers the host, so no new registrable domain is added.
2150        let arxiv = tier_1_allowlist()
2151            .into_iter()
2152            .find(|a| a.source == "arxiv")
2153            .expect("arxiv entry");
2154        assert!(
2155            arxiv.matches("ar5iv.labs.arxiv.org"),
2156            "ar5iv host must fall under the existing *.arxiv.org surface"
2157        );
2158    }
2159
2160    #[test]
2161    fn oa_publisher_allowlist_groups_under_one_synthetic_source() {
2162        // The OA-publisher fan-out from Unpaywall's `best_oa_location.url`
2163        // is keyed under a single synthetic `"oa-publisher"` source so the
2164        // orchestrator can pass that one source key to
2165        // `HttpClient::fetch_pdf`. See `docs/REDIRECT_ALLOWLIST.md` §3 (the
2166        // informed-best-effort note) and the function-level docs in
2167        // [`oa_publisher_allowlist`].
2168        let lists = oa_publisher_allowlist();
2169        assert_eq!(lists.len(), 1, "exactly one synthetic source entry");
2170        assert_eq!(lists[0].source, "oa-publisher");
2171    }
2172
2173    #[test]
2174    fn oa_publisher_allowlist_matches_known_oa_hosts() {
2175        let lists = oa_publisher_allowlist();
2176        let oa = lists
2177            .iter()
2178            .find(|a| a.source == "oa-publisher")
2179            .expect("oa-publisher entry");
2180        // Spot-check a representative entry per host family.
2181        assert!(oa.matches("link.springer.com"));
2182        assert!(oa.matches("nature.com"));
2183        assert!(oa.matches("onlinelibrary.wiley.com"));
2184        assert!(oa.matches("www.frontiersin.org"));
2185        assert!(oa.matches("www.mdpi.com"));
2186        assert!(oa.matches("journals.plos.org"));
2187        assert!(oa.matches("www.biorxiv.org"));
2188        assert!(oa.matches("europepmc.org"));
2189        assert!(oa.matches("www.ncbi.nlm.nih.gov"));
2190        assert!(oa.matches("arxiv.org"));
2191        // #193: physics-society / diamond-OA hosts (empirically observed
2192        // as Unpaywall best_oa_location targets in the dogfood run).
2193        assert!(oa.matches("link.aps.org"));
2194        assert!(oa.matches("journals.aps.org"));
2195        assert!(oa.matches("scipost.org"));
2196        assert!(oa.matches("www.scipost.org"));
2197        assert!(oa.matches("iopscience.iop.org"));
2198        // Document intent of the `*.<suffix>` form: per
2199        // `REDIRECT_ALLOWLIST.md` §2.2 rule 3 it matches the bare
2200        // registrable domain AND any subdomain. Unpaywall has not been
2201        // observed returning bare-domain PDF URLs for these publishers,
2202        // but accepting them is consistent with every other `*.` entry in
2203        // this list (e.g. `arxiv.org` matched by `*.arxiv.org`) and is
2204        // what the matching rule already implements.
2205        assert!(oa.matches("aps.org"));
2206        assert!(oa.matches("iop.org"));
2207        // Multi-level subdomains also match (e.g. SciPost's deep paths);
2208        // documents the wildcard scope rather than testing a known URL.
2209        assert!(oa.matches("submissions.scipost.org"));
2210        // Negative: an attacker host is not covered.
2211        assert!(!oa.matches("attacker.test"));
2212        // Negative: dot-boundary safety for the new entries — a different
2213        // suffix that merely ends with the registrable name must NOT match.
2214        assert!(!oa.matches("notaps.org"));
2215        assert!(!oa.matches("evilscipost.org"));
2216        assert!(!oa.matches("notiop.org"));
2217        // Negative: dot-boundary safety — `*.springer.com` must not match
2218        // `notspringer.com`.
2219        assert!(!oa.matches("notspringer.com"));
2220    }
2221
2222    #[test]
2223    fn allowlist_matches_exact_fqdn() {
2224        let a = SourceAllowlist::new("crossref", vec!["api.crossref.org".to_string()]);
2225        assert!(a.matches("api.crossref.org"));
2226        assert!(!a.matches("crossref.org"));
2227        assert!(!a.matches("xapi.crossref.org"));
2228    }
2229
2230    #[test]
2231    fn allowlist_matches_subdomain_glob() {
2232        // Per docs/REDIRECT_ALLOWLIST.md §2.2 rule 3: `*.<suffix>`
2233        // matches both `<suffix>` itself AND any `*.<suffix>` subdomain,
2234        // but never matches a different suffix that happens to end with
2235        // `<suffix>` without a dot boundary.
2236        let a = SourceAllowlist::new("crossref", vec!["*.crossref.org".to_string()]);
2237        assert!(a.matches("doi.crossref.org"));
2238        assert!(a.matches("crossref.org"));
2239        assert!(!a.matches("notcrossref.org"));
2240        assert!(!a.matches("crossref.org.attacker.test"));
2241    }
2242
2243    #[test]
2244    fn allowlist_matches_is_case_insensitive() {
2245        let a = SourceAllowlist::new("crossref", vec!["API.crossref.ORG".to_string()]);
2246        assert!(a.matches("api.crossref.org"));
2247        assert!(a.matches("API.CROSSREF.ORG"));
2248    }
2249
2250    #[test]
2251    fn allowlist_with_no_redirect_hosts_matches_nothing() {
2252        // §2.2 rule 5: an empty `redirect_hosts` means "no redirects
2253        // permitted from this source".
2254        let a = SourceAllowlist::new("ghost", Vec::<String>::new());
2255        assert!(!a.matches("anything.test"));
2256        assert!(!a.matches(""));
2257    }
2258
2259    // ---------------------------------------------------------------
2260    // PDF magic-byte handling — tests on the body-parsing path. We
2261    // exercise the magic-byte branch via the public API against a
2262    // wiremock server so the assertion runs through the full
2263    // streaming codepath.
2264    // ---------------------------------------------------------------
2265
2266    /// Build a test-only `HttpClient` against an `http://` wiremock
2267    /// origin.
2268    ///
2269    /// Slice 5 (PR #84 advisory item A4 refactor): this helper now
2270    /// delegates to the public
2271    /// [`HttpClient::new_for_tests_allow_http`] constructor (defined
2272    /// just above the test module) instead of re-implementing the
2273    /// redirect-policy + `https_only(false)` builder. The two
2274    /// implementations had drifted into duplicates — keeping a private
2275    /// re-implementation only meant a future security tweak to the
2276    /// builder would silently leave the tests on a stale path.
2277    fn build_test_client_for_http(source: &str, allowlist_host: &str) -> HttpClient {
2278        HttpClient::new_for_tests_allow_http(source, allowlist_host)
2279    }
2280
2281    #[tokio::test]
2282    async fn pdf_magic_byte_match_succeeds() {
2283        let server = MockServer::start().await;
2284        let body = b"%PDF-1.7\n...some pdf bytes...".to_vec();
2285        Mock::given(method("GET"))
2286            .and(path("/paper.pdf"))
2287            .respond_with(ResponseTemplate::new(200).set_body_bytes(body.clone()))
2288            .mount(&server)
2289            .await;
2290        let host = server
2291            .uri()
2292            .parse::<Url>()
2293            .unwrap()
2294            .host_str()
2295            .unwrap()
2296            .to_string();
2297        let client = build_test_client_for_http("crossref", &host);
2298        let url: Url = format!("{}/paper.pdf", server.uri()).parse().unwrap();
2299        let (got_body, _final_url) = client.fetch_pdf("crossref", url).await.expect("ok");
2300        assert_eq!(&got_body[..], &body[..]);
2301    }
2302
2303    #[tokio::test]
2304    async fn pdf_magic_byte_mismatch_rejects() {
2305        let server = MockServer::start().await;
2306        Mock::given(method("GET"))
2307            .and(path("/not_a_pdf"))
2308            .respond_with(
2309                ResponseTemplate::new(200).set_body_bytes(b"<html>not a pdf</html>".to_vec()),
2310            )
2311            .mount(&server)
2312            .await;
2313        let host = server
2314            .uri()
2315            .parse::<Url>()
2316            .unwrap()
2317            .host_str()
2318            .unwrap()
2319            .to_string();
2320        let client = build_test_client_for_http("crossref", &host);
2321        let url: Url = format!("{}/not_a_pdf", server.uri()).parse().unwrap();
2322        let err = client
2323            .fetch_pdf("crossref", url)
2324            .await
2325            .expect_err("not pdf");
2326        match err {
2327            HttpError::NotAPdf { got } => {
2328                assert_eq!(&got, b"<html");
2329            }
2330            other => panic!("expected NotAPdf, got {:?}", other),
2331        }
2332    }
2333
2334    #[tokio::test]
2335    async fn fetch_bytes_does_not_check_pdf_magic() {
2336        // The non-PDF path returns the body unchanged regardless of
2337        // magic bytes. This pins the boundary between the JSON/text
2338        // path and the PDF path.
2339        let server = MockServer::start().await;
2340        Mock::given(method("GET"))
2341            .and(path("/data.json"))
2342            .respond_with(
2343                ResponseTemplate::new(200).set_body_bytes(br#"{"hello":"world"}"#.to_vec()),
2344            )
2345            .mount(&server)
2346            .await;
2347        let host = server
2348            .uri()
2349            .parse::<Url>()
2350            .unwrap()
2351            .host_str()
2352            .unwrap()
2353            .to_string();
2354        let client = build_test_client_for_http("crossref", &host);
2355        let url: Url = format!("{}/data.json", server.uri()).parse().unwrap();
2356        let (body, _final_url) = client.fetch_bytes("crossref", url).await.expect("ok");
2357        assert_eq!(&body[..], br#"{"hello":"world"}"#);
2358    }
2359
2360    #[tokio::test]
2361    async fn oversized_body_via_content_length_short_circuits() {
2362        // Wiremock can advertise a `Content-Length` larger than the body
2363        // it actually serves; hyper accepts the mismatch and our
2364        // fast-path check fires before any body bytes are consumed.
2365        let server = MockServer::start().await;
2366        let oversized = PDF_MAX_BYTES + 1;
2367        Mock::given(method("GET"))
2368            .and(path("/huge"))
2369            .respond_with(
2370                ResponseTemplate::new(200)
2371                    .insert_header("content-length", oversized.to_string().as_str())
2372                    .set_body_bytes(b"%PDF-".to_vec()),
2373            )
2374            .mount(&server)
2375            .await;
2376        let host = server
2377            .uri()
2378            .parse::<Url>()
2379            .unwrap()
2380            .host_str()
2381            .unwrap()
2382            .to_string();
2383        let client = build_test_client_for_http("crossref", &host);
2384        let url: Url = format!("{}/huge", server.uri()).parse().unwrap();
2385        let err = client
2386            .fetch_bytes("crossref", url)
2387            .await
2388            .expect_err("should reject");
2389        match err {
2390            HttpError::OversizedBody { actual, cap } => {
2391                assert!(actual > cap, "actual {} should exceed cap {}", actual, cap);
2392                assert_eq!(cap, PDF_MAX_BYTES);
2393            }
2394            // The mismatched Content-Length may also trip an underlying
2395            // transport error before our fast-path runs. Either outcome
2396            // satisfies the security goal (the transfer was aborted
2397            // without buffering 100 GB), so accept Network here as a
2398            // wiremock idiosyncrasy rather than a contract relaxation.
2399            HttpError::Network(_) => {}
2400            other => panic!("expected OversizedBody or Network, got {:?}", other),
2401        }
2402    }
2403
2404    #[tokio::test]
2405    async fn unknown_source_rejected() {
2406        let client = HttpClient::new(tier_1_allowlist()).expect("client builds");
2407        let url: Url = "https://api.crossref.org/works/10.1234/x".parse().unwrap();
2408        let err = client
2409            .fetch_bytes("not-a-source", url)
2410            .await
2411            .expect_err("unknown source");
2412        match err {
2413            HttpError::UnknownSource { source_key } => {
2414                assert_eq!(source_key, "not-a-source")
2415            }
2416            other => panic!("expected UnknownSource, got {:?}", other),
2417        }
2418    }
2419
2420    #[tokio::test]
2421    async fn http_status_error_surfaces() {
2422        let server = MockServer::start().await;
2423        Mock::given(method("GET"))
2424            .and(path("/missing"))
2425            .respond_with(ResponseTemplate::new(404))
2426            .mount(&server)
2427            .await;
2428        let host = server
2429            .uri()
2430            .parse::<Url>()
2431            .unwrap()
2432            .host_str()
2433            .unwrap()
2434            .to_string();
2435        let client = build_test_client_for_http("crossref", &host);
2436        let url: Url = format!("{}/missing", server.uri()).parse().unwrap();
2437        let err = client.fetch_bytes("crossref", url).await.expect_err("404");
2438        match err {
2439            HttpError::HttpStatus { status, .. } => assert_eq!(status, 404),
2440            other => panic!("expected HttpStatus, got {:?}", other),
2441        }
2442    }
2443
2444    // ---------------------------------------------------------------
2445    // Redirect policy tests — drive the closure via wiremock 30x
2446    // responses pointing at insecure / off-allowlist targets. With
2447    // `https_only(true)` on the production builder the request never
2448    // leaves the initial leg — we run these against the test builder
2449    // (which relaxes `https_only` for the *initial* leg only) so the
2450    // redirect closure is reached and exercised.
2451    // ---------------------------------------------------------------
2452
2453    #[tokio::test]
2454    async fn redirect_to_http_is_rejected_by_closure() {
2455        let server = MockServer::start().await;
2456        Mock::given(method("GET"))
2457            .and(path("/redir"))
2458            .respond_with(
2459                ResponseTemplate::new(302).insert_header("location", "http://attacker.test/file"),
2460            )
2461            .mount(&server)
2462            .await;
2463        let host = server
2464            .uri()
2465            .parse::<Url>()
2466            .unwrap()
2467            .host_str()
2468            .unwrap()
2469            .to_string();
2470        let client = build_test_client_for_http("crossref", &host);
2471        let url: Url = format!("{}/redir", server.uri()).parse().unwrap();
2472        let err = client
2473            .fetch_bytes("crossref", url)
2474            .await
2475            .expect_err("redirect to http rejected");
2476        match err {
2477            HttpError::Network(e) => {
2478                let msg = format!("{:?}", e);
2479                assert!(
2480                    msg.contains("InsecureRedirect") || msg.contains("non-HTTPS"),
2481                    "expected insecure-redirect signal in error chain, got {}",
2482                    msg
2483                );
2484            }
2485            other => panic!("expected Network(InsecureRedirect), got {:?}", other),
2486        }
2487    }
2488
2489    #[tokio::test]
2490    async fn redirect_outside_allowlist_is_rejected_by_closure() {
2491        let server = MockServer::start().await;
2492        Mock::given(method("GET"))
2493            .and(path("/redir"))
2494            .respond_with(
2495                ResponseTemplate::new(302).insert_header("location", "https://attacker.test/file"),
2496            )
2497            .mount(&server)
2498            .await;
2499        let host = server
2500            .uri()
2501            .parse::<Url>()
2502            .unwrap()
2503            .host_str()
2504            .unwrap()
2505            .to_string();
2506        let client = build_test_client_for_http("crossref", &host);
2507        let url: Url = format!("{}/redir", server.uri()).parse().unwrap();
2508        let err = client
2509            .fetch_bytes("crossref", url)
2510            .await
2511            .expect_err("redirect to attacker rejected");
2512        match err {
2513            HttpError::Network(e) => {
2514                let msg = format!("{:?}", e);
2515                assert!(
2516                    msg.contains("RedirectDenied") || msg.contains("not in allowlist"),
2517                    "expected redirect-denied signal in error chain, got {}",
2518                    msg
2519                );
2520            }
2521            other => panic!("expected Network(RedirectDenied), got {:?}", other),
2522        }
2523    }
2524
2525    #[tokio::test]
2526    async fn redirect_to_allowlisted_https_host_is_followed_by_closure() {
2527        // 302 to an https host that IS in the allowlist. The redirect
2528        // dispatch will fail (DNS won't resolve `mirror.allowed.test`)
2529        // but the closure must NOT short-circuit — failure mode is a
2530        // transport error, not InsecureRedirect / RedirectDenied.
2531        let server = MockServer::start().await;
2532        Mock::given(method("GET"))
2533            .and(path("/redir"))
2534            .respond_with(
2535                ResponseTemplate::new(302)
2536                    .insert_header("location", "https://mirror.allowed.test/file"),
2537            )
2538            .mount(&server)
2539            .await;
2540        let initial_host = server
2541            .uri()
2542            .parse::<Url>()
2543            .unwrap()
2544            .host_str()
2545            .unwrap()
2546            .to_string();
2547        // Allow the initial host AND the redirect target host.
2548        let allowlist = SourceAllowlist::new(
2549            "crossref",
2550            vec![initial_host.clone(), "*.allowed.test".to_string()],
2551        );
2552        let allowlist_for_closure = allowlist.clone();
2553        let policy = Policy::custom(move |attempt| {
2554            let scheme = attempt.url().scheme().to_string();
2555            let host_opt = attempt.url().host_str().map(|h| h.to_ascii_lowercase());
2556            if scheme != "https" {
2557                return attempt.error(HttpError::InsecureRedirect { scheme });
2558            }
2559            let h = match host_opt {
2560                Some(h) => h,
2561                None => {
2562                    return attempt.error(HttpError::RedirectDenied {
2563                        source_key: allowlist_for_closure.source.clone(),
2564                        host: String::new(),
2565                        expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
2566                    });
2567                }
2568            };
2569            // `permits`, mirroring production: this closure is a local copy
2570            // of `build_client`'s policy, and a copy that adjudicates
2571            // differently tests something that does not ship (#533).
2572            if !allowlist_for_closure.permits(&h) {
2573                return attempt.error(HttpError::RedirectDenied {
2574                    source_key: allowlist_for_closure.source.clone(),
2575                    host: h,
2576                    expected_hosts: allowlist_for_closure.redirect_hosts.clone(),
2577                });
2578            }
2579            attempt.follow()
2580        });
2581        ensure_crypto_provider();
2582        let raw_client = ClientBuilder::new()
2583            .https_only(false)
2584            .redirect(policy)
2585            .connect_timeout(CONNECT_TIMEOUT)
2586            .timeout(Duration::from_secs(5))
2587            .user_agent("doiget/test")
2588            .tls_backend_rustls()
2589            .build()
2590            .expect("client builds");
2591        let url: Url = format!("{}/redir", server.uri()).parse().unwrap();
2592        let err = raw_client.get(url).send().await.expect_err("DNS fails");
2593        // The error should NOT carry our InsecureRedirect / RedirectDenied
2594        // marker — the closure approved the redirect.
2595        let msg = format!("{:?}", err);
2596        assert!(
2597            !msg.contains("RedirectDenied") && !msg.contains("InsecureRedirect"),
2598            "closure short-circuited an allowed redirect: {}",
2599            msg,
2600        );
2601    }
2602
2603    #[test]
2604    fn http_client_clone_is_cheap() {
2605        // Sanity: cloning shares the inner Arc<HashMap<...>>.
2606        let a = HttpClient::new(tier_1_allowlist()).expect("builds");
2607        let b = a.clone();
2608        assert_eq!(a.clients.len(), b.clients.len());
2609        assert!(Arc::ptr_eq(&a.clients, &b.clients));
2610    }
2611
2612    // ---------------------------------------------------------------
2613    // HttpError -> Option<DenialContext>  (ADR-0023 §4 mapping)
2614    // ---------------------------------------------------------------
2615
2616    #[test]
2617    fn denial_from_redirect_denied_carries_attempted_and_expected() {
2618        use crate::{DenialContext, DenialReason};
2619        let e = HttpError::RedirectDenied {
2620            source_key: "crossref".to_string(),
2621            host: "evil.example.com".to_string(),
2622            expected_hosts: vec!["api.crossref.org".to_string(), "*.crossref.org".to_string()],
2623        };
2624        let dc: Option<DenialContext> = (&e).into();
2625        let dc = dc.expect("RedirectDenied -> Some(DenialContext)");
2626        assert_eq!(dc.reason, DenialReason::RedirectNotInAllowlist);
2627        assert_eq!(dc.source.as_deref(), Some("crossref"));
2628        assert_eq!(dc.attempted.as_deref(), Some("evil.example.com"));
2629        assert_eq!(
2630            dc.expected.as_deref(),
2631            Some(&["api.crossref.org".to_string(), "*.crossref.org".to_string()][..])
2632        );
2633        assert!(dc.cap.is_none());
2634        assert!(dc.actual.is_none());
2635        assert!(dc.hop_index.is_none());
2636    }
2637
2638    #[test]
2639    fn denial_from_oversized_body_carries_cap_and_actual() {
2640        use crate::{DenialContext, DenialReason};
2641        let e = HttpError::OversizedBody {
2642            actual: 209_715_200,
2643            cap: PDF_MAX_BYTES,
2644        };
2645        let dc: Option<DenialContext> = (&e).into();
2646        let dc = dc.expect("OversizedBody -> Some(DenialContext)");
2647        assert_eq!(dc.reason, DenialReason::SizeCapExceeded);
2648        assert_eq!(dc.cap, Some(PDF_MAX_BYTES));
2649        assert_eq!(dc.actual, Some(209_715_200));
2650        assert!(dc.source.is_none());
2651        assert!(dc.attempted.is_none());
2652        // OversizedBody has no allowlist channel: producer leaves
2653        // `expected` at `None` (NOT `Some(vec![])`). See the field doc on
2654        // `DenialContext::expected` for the disambiguation.
2655        assert!(dc.expected.is_none());
2656    }
2657
2658    #[test]
2659    fn denial_from_not_a_pdf_hex_encodes_got_bytes() {
2660        use crate::{DenialContext, DenialReason};
2661        // First 5 bytes of "<html" — what the magic-byte check sees when
2662        // a publisher returns an HTML interstitial instead of a PDF.
2663        let e = HttpError::NotAPdf {
2664            got: [0x3c, 0x68, 0x74, 0x6d, 0x6c],
2665        };
2666        let dc: Option<DenialContext> = (&e).into();
2667        let dc = dc.expect("NotAPdf -> Some(DenialContext)");
2668        assert_eq!(dc.reason, DenialReason::ContentTypeMismatch);
2669        assert_eq!(dc.attempted.as_deref(), Some("3c68746d6c"));
2670        assert_eq!(dc.expected.as_deref(), Some(&["%PDF-".to_string()][..]));
2671    }
2672
2673    #[test]
2674    fn denial_from_insecure_redirect_marks_insecure_scheme() {
2675        use crate::{DenialContext, DenialReason};
2676        let e = HttpError::InsecureRedirect {
2677            scheme: "http".to_string(),
2678        };
2679        let dc: Option<DenialContext> = (&e).into();
2680        let dc = dc.expect("InsecureRedirect -> Some(DenialContext)");
2681        // ADR-0023 §4 (post-incorporation review): InsecureRedirect maps
2682        // to its own dedicated `InsecureScheme` reason, not the host-
2683        // allowlist reason — they are semantically distinct denials.
2684        assert_eq!(dc.reason, DenialReason::InsecureScheme);
2685        assert_eq!(dc.attempted.as_deref(), Some("http:..."));
2686        assert_eq!(dc.expected.as_deref(), Some(&["https".to_string()][..]));
2687    }
2688
2689    #[test]
2690    fn denial_from_non_denial_variants_returns_none() {
2691        use crate::DenialContext;
2692        // Network / HttpStatus / UnknownSource are not denials; they
2693        // map to None per ADR-0023 §4.
2694        let e = HttpError::HttpStatus {
2695            status: 503,
2696            retry_after_ms: None,
2697            url: "https://api.crossref.org/works/x".to_string(),
2698        };
2699        let dc: Option<DenialContext> = (&e).into();
2700        assert!(dc.is_none(), "HttpStatus must not produce a DenialContext");
2701
2702        let e = HttpError::UnknownSource {
2703            source_key: "ghost".to_string(),
2704        };
2705        let dc: Option<DenialContext> = (&e).into();
2706        assert!(
2707            dc.is_none(),
2708            "UnknownSource must not produce a DenialContext"
2709        );
2710    }
2711
2712    // ---------------------------------------------------------------
2713    // Issue #117 — transient retry / backoff. Real time: wiremock
2714    // serves over real localhost IO and tokio `start_paused` is
2715    // incompatible with that (it auto-advances past reqwest's
2716    // timeout). Backoff is small enough that the slowest case
2717    // (persistent 503, 3 retries ≈ 3.5s) stays within the suite budget.
2718    // ---------------------------------------------------------------
2719
2720    fn host_of(server: &MockServer) -> String {
2721        server
2722            .uri()
2723            .parse::<Url>()
2724            .unwrap()
2725            .host_str()
2726            .unwrap()
2727            .to_string()
2728    }
2729
2730    #[tokio::test]
2731    async fn transient_503_then_200_succeeds() {
2732        let server = MockServer::start().await;
2733        // Catch-all 200 mounted first (lowest precedence); the
2734        // single-shot 503 mounted last takes precedence for the first
2735        // request only, then falls through to the 200.
2736        Mock::given(method("GET"))
2737            .and(path("/p"))
2738            .respond_with(ResponseTemplate::new(200).set_body_string(r#"{"ok":1}"#))
2739            .mount(&server)
2740            .await;
2741        Mock::given(method("GET"))
2742            .and(path("/p"))
2743            .respond_with(ResponseTemplate::new(503))
2744            .up_to_n_times(1)
2745            .mount(&server)
2746            .await;
2747
2748        let client = build_test_client_for_http("crossref", &host_of(&server));
2749        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2750        let (body, _) = client
2751            .fetch_bytes("crossref", url)
2752            .await
2753            .expect("503-then-200 must succeed after one retry");
2754        assert_eq!(&body[..], br#"{"ok":1}"#);
2755    }
2756
2757    #[tokio::test]
2758    async fn persistent_503_exhausts_and_returns_httpstatus() {
2759        let server = MockServer::start().await;
2760        Mock::given(method("GET"))
2761            .and(path("/p"))
2762            .respond_with(ResponseTemplate::new(503))
2763            .mount(&server)
2764            .await;
2765
2766        let client = build_test_client_for_http("crossref", &host_of(&server));
2767        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2768        let err = client
2769            .fetch_bytes("crossref", url)
2770            .await
2771            .expect_err("persistent 503 must exhaust retries");
2772        match err {
2773            HttpError::HttpStatus { status, .. } => assert_eq!(status, 503),
2774            other => panic!("expected HttpStatus 503, got {other:?}"),
2775        }
2776        // First attempt + MAX_FETCH_RETRIES retries.
2777        let reqs = server
2778            .received_requests()
2779            .await
2780            .expect("wiremock records requests");
2781        assert_eq!(reqs.len(), (MAX_FETCH_RETRIES + 1) as usize);
2782    }
2783
2784    #[tokio::test]
2785    async fn retry_after_429_then_200_succeeds() {
2786        let server = MockServer::start().await;
2787        Mock::given(method("GET"))
2788            .and(path("/p"))
2789            .respond_with(ResponseTemplate::new(200).set_body_string("ok"))
2790            .mount(&server)
2791            .await;
2792        Mock::given(method("GET"))
2793            .and(path("/p"))
2794            .respond_with(ResponseTemplate::new(429).insert_header("Retry-After", "1"))
2795            .up_to_n_times(1)
2796            .mount(&server)
2797            .await;
2798
2799        let client = build_test_client_for_http("crossref", &host_of(&server));
2800        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2801        let (body, _) = client
2802            .fetch_bytes("crossref", url)
2803            .await
2804            .expect("429+Retry-After then 200 must succeed");
2805        assert_eq!(&body[..], b"ok");
2806    }
2807
2808    /// #506: the server's own `Retry-After` survives to the caller.
2809    ///
2810    /// `parse_retry_after` already read this header, but only on the RETRY
2811    /// path -- the terminal `return` discarded it, so by the time the error
2812    /// reached a caller the number was gone and `error.retry_after_ms` looked
2813    /// impossible to fill honestly. It is not: the response that ends the
2814    /// attempt carries its own header, and that is the one to wait.
2815    #[tokio::test]
2816    async fn a_terminal_429_carries_the_servers_retry_after() {
2817        let server = MockServer::start().await;
2818        // 429 on every attempt, so the retries are exhausted and the error is
2819        // the terminal one -- the case that used to lose the header.
2820        Mock::given(method("GET"))
2821            .and(path("/p"))
2822            .respond_with(ResponseTemplate::new(429).insert_header("Retry-After", "7"))
2823            .mount(&server)
2824            .await;
2825
2826        let client = build_test_client_for_http("crossref", &host_of(&server));
2827        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2828        let err = client
2829            .fetch_bytes("crossref", url)
2830            .await
2831            .expect_err("every attempt 429s");
2832
2833        match err {
2834            HttpError::HttpStatus {
2835                status,
2836                retry_after_ms,
2837                ..
2838            } => {
2839                assert_eq!(status, 429);
2840                assert_eq!(
2841                    retry_after_ms,
2842                    Some(7_000),
2843                    "the SERVER's number, in ms, not our backoff"
2844                );
2845            }
2846            other => panic!("expected HttpStatus, got {other:?}"),
2847        }
2848    }
2849
2850    /// No header, no number. Backfilling from `backoff_delay` would hand the
2851    /// caller a guess about the server wearing the name of a value the server
2852    /// supplied -- the defect this field exists to avoid.
2853    #[tokio::test]
2854    async fn a_terminal_429_without_the_header_carries_no_number() {
2855        let server = MockServer::start().await;
2856        Mock::given(method("GET"))
2857            .and(path("/p"))
2858            .respond_with(ResponseTemplate::new(429))
2859            .mount(&server)
2860            .await;
2861
2862        let client = build_test_client_for_http("crossref", &host_of(&server));
2863        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2864        let err = client
2865            .fetch_bytes("crossref", url)
2866            .await
2867            .expect_err("every attempt 429s");
2868
2869        match err {
2870            HttpError::HttpStatus { retry_after_ms, .. } => assert_eq!(retry_after_ms, None),
2871            other => panic!("expected HttpStatus, got {other:?}"),
2872        }
2873    }
2874
2875    #[tokio::test]
2876    async fn permanent_404_is_not_retried() {
2877        let server = MockServer::start().await;
2878        Mock::given(method("GET"))
2879            .and(path("/p"))
2880            .respond_with(ResponseTemplate::new(404))
2881            .mount(&server)
2882            .await;
2883
2884        let client = build_test_client_for_http("crossref", &host_of(&server));
2885        let url: Url = format!("{}/p", server.uri()).parse().unwrap();
2886        let _ = client
2887            .fetch_bytes("crossref", url)
2888            .await
2889            .expect_err("404 must fail");
2890        let reqs = server
2891            .received_requests()
2892            .await
2893            .expect("wiremock records requests");
2894        assert_eq!(reqs.len(), 1, "4xx (non-408/429) must NOT be retried");
2895    }
2896}