Skip to main content

doiget_core/
preprint.rs

1//! Finding a DOI's arXiv preprint when Unpaywall does not name one (#636,
2//! ADR-0062).
3//!
4//! The #325 fallback fetches an arXiv preprint in place of a blocked or
5//! missing publisher copy, but only one Unpaywall reports. A paper can be
6//! closed at its publisher, unknown to Unpaywall, and on arXiv all the same
7//! (measured on 10.1103/bbnt-brjz, arXiv:2512.07923). Three ways to find it,
8//! tried in order, each stopping the search when it answers:
9//!
10//! 1. **Crossref `relation.has-preprint`** -- in the Crossref record the
11//!    fetch already holds, so it costs no request. An `arxiv` id, or an
12//!    arXiv DOI (`10.48550/arXiv.<id>`).
13//! 2. **OpenAlex `locations[]`** -- a location whose landing page is
14//!    `arxiv.org/abs/<id>`. One request to `api.openalex.org`, and only when
15//!    the user enabled OpenAlex (`DOIGET_ENABLE_OPENALEX`), so a default
16//!    fetch contacts no host it did not before.
17//! 3. **arXiv search by title and first author** -- one request to arXiv's
18//!    API, at arXiv's own rate. A hit counts only when its title is the
19//!    record's title (compared letters and digits only, case-folded), its
20//!    authors include the first author's surname, and it does not name a
21//!    different published DOI.
22//!
23//! Each answer says which of the three found it; nothing here fetches.
24
25use serde_json::Value;
26use url::Url;
27
28use crate::source::{FetchContext, FetchError};
29
30/// The user's own NASA ADS token (#644). Never shipped, never logged.
31pub const ADS_TOKEN_ENV: &str = "DOIGET_ADS_TOKEN";
32use crate::{ArxivId, Doi};
33
34/// Which method found a preprint.
35#[derive(Debug, Clone, Copy, PartialEq, Eq)]
36pub enum FoundBy {
37    /// Unpaywall named an arXiv location for the DOI itself (#325).
38    Unpaywall,
39    /// Crossref's `relation.has-preprint`.
40    CrossrefRelation,
41    /// An OpenAlex location on arXiv.
42    OpenAlexLocation,
43    /// arXiv's search, by title and first author.
44    ArxivTitleSearch,
45    /// bioRxiv / medRxiv's `pubs` endpoint (#640).
46    BiorxivPubs,
47    /// INSPIRE-HEP's `arxiv_eprints` for the DOI (#642).
48    Inspire,
49    /// NASA ADS's `identifier` for the DOI (#644).
50    Ads,
51}
52
53impl FoundBy {
54    /// Wire token.
55    #[must_use]
56    pub const fn as_str(self) -> &'static str {
57        match self {
58            Self::Unpaywall => "unpaywall",
59            Self::CrossrefRelation => "crossref_relation",
60            Self::OpenAlexLocation => "openalex_location",
61            Self::ArxivTitleSearch => "arxiv_title_search",
62            Self::BiorxivPubs => "biorxiv_pubs",
63            Self::Inspire => "inspire",
64            Self::Ads => "ads",
65        }
66    }
67}
68
69/// A preprint found for a DOI.
70#[derive(Debug, Clone, PartialEq, Eq)]
71pub struct Found {
72    /// The arXiv id, version suffix removed.
73    pub arxiv_id: ArxivId,
74    /// How it was found.
75    pub found_by: FoundBy,
76}
77
78/// The arXiv id Crossref's `relation.has-preprint` names, if any.
79#[must_use]
80pub fn from_crossref(crossref_message: &Value) -> Option<ArxivId> {
81    let rels = crossref_message
82        .pointer("/relation/has-preprint")?
83        .as_array()?;
84    rels.iter().find_map(|r| {
85        let id = r.get("id").and_then(Value::as_str)?.trim();
86        match r.get("id-type").and_then(Value::as_str)? {
87            "arxiv" => arxiv_id(id),
88            "doi" => {
89                let lower = id.to_ascii_lowercase();
90                lower
91                    .strip_prefix("10.48550/arxiv.")
92                    .and_then(|_| arxiv_id(&id["10.48550/arxiv.".len()..]))
93            }
94            _ => None,
95        }
96    })
97}
98
99/// A non-arXiv preprint found for a DOI (#640): its own DOI, fetched through
100/// the ordinary OA route (the location Unpaywall reports for it).
101#[derive(Debug, Clone, PartialEq, Eq)]
102pub struct FoundDoi {
103    /// The preprint's DOI.
104    pub doi: Doi,
105    /// The platform, when the finder named one (`bioRxiv`, `medRxiv`).
106    pub platform: Option<String>,
107    /// How it was found.
108    pub found_by: FoundBy,
109}
110
111/// A non-arXiv preprint DOI Crossref's `relation.has-preprint` names: any
112/// DOI but an arXiv one (`10.48550`), which [`from_crossref`] handles.
113#[must_use]
114pub fn preprint_doi_from_crossref(crossref_message: &Value) -> Option<Doi> {
115    let rels = crossref_message
116        .pointer("/relation/has-preprint")?
117        .as_array()?;
118    rels.iter().find_map(|r| {
119        (r.get("id-type").and_then(Value::as_str)? == "doi")
120            .then(|| r.get("id").and_then(Value::as_str))
121            .flatten()
122            .map(str::trim)
123            .filter(|id| !id.to_ascii_lowercase().starts_with("10.48550/"))
124            .and_then(|id| Doi::parse(id).ok())
125    })
126}
127
128/// The preprint DOI a bioRxiv / medRxiv `pubs` answer names.
129fn from_pubs(answer: &Value) -> Option<(Doi, Option<String>)> {
130    let first = answer.get("collection")?.as_array()?.first()?;
131    let doi = Doi::parse(first.get("preprint_doi")?.as_str()?.trim()).ok()?;
132    let platform = first
133        .get("preprint_platform")
134        .and_then(Value::as_str)
135        .map(str::to_string);
136    Some((doi, platform))
137}
138
139/// Look for a non-arXiv preprint of `doi` (#640): Crossref's relation, then
140/// bioRxiv and medRxiv `pubs` when `biorxiv_enabled`.
141///
142/// # Errors
143///
144/// A provenance-log failure only; a request that fails is a finder that
145/// found nothing.
146pub async fn find_preprint_doi(
147    doi: &Doi,
148    crossref_message: &Value,
149    biorxiv_enabled: bool,
150    ctx: &FetchContext,
151) -> Result<Option<FoundDoi>, FetchError> {
152    if let Some(found) = preprint_doi_from_crossref(crossref_message) {
153        return Ok(Some(FoundDoi {
154            doi: found,
155            platform: None,
156            found_by: FoundBy::CrossrefRelation,
157        }));
158    }
159    if !biorxiv_enabled {
160        return Ok(None);
161    }
162    for server in ["biorxiv", "medrxiv"] {
163        let mut url = base("DOIGET_BIORXIV_BASE", "https://api.biorxiv.org")?;
164        url.set_path(&format!("/pubs/{server}/{}/na/json", doi.as_str()));
165        match logged_get(doi, "biorxiv", url, ctx).await {
166            Ok(Some(body)) => {
167                if let Some((found, platform)) = serde_json::from_slice::<Value>(&body)
168                    .ok()
169                    .as_ref()
170                    .and_then(from_pubs)
171                {
172                    return Ok(Some(FoundDoi {
173                        doi: found,
174                        platform,
175                        found_by: FoundBy::BiorxivPubs,
176                    }));
177                }
178            }
179            Ok(None) => {}
180            Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
181            Err(e) => {
182                tracing::info!(error = %e, server, "preprint lookup: bioRxiv pubs did not answer")
183            }
184        }
185    }
186    Ok(None)
187}
188
189/// The arXiv id of an OpenAlex work's arXiv location, if any.
190#[must_use]
191pub fn from_openalex_work(work: &Value) -> Option<ArxivId> {
192    work.get("locations")?.as_array()?.iter().find_map(|l| {
193        let url = l.get("landing_page_url").and_then(Value::as_str)?;
194        let rest = url
195            .strip_prefix("http://arxiv.org/abs/")
196            .or_else(|| url.strip_prefix("https://arxiv.org/abs/"))?;
197        arxiv_id(rest)
198    })
199}
200
201/// The arXiv id INSPIRE-HEP's record gives (`metadata.arxiv_eprints`), if
202/// any. Nothing else in the record is read (#642): its `documents` are files
203/// whose provenance and licence are not stated per file.
204#[must_use]
205pub fn from_inspire_record(record: &Value) -> Option<ArxivId> {
206    record
207        .pointer("/metadata/arxiv_eprints")?
208        .as_array()?
209        .iter()
210        .find_map(|e| e.get("value").and_then(Value::as_str).and_then(arxiv_id))
211}
212
213/// The arXiv id in an ADS search answer's first record: the
214/// `identifier` entry `arXiv:<id>` (ADS's search syntax lists arXiv ids
215/// among a record's identifiers). Nothing else is read (#644).
216#[must_use]
217pub fn from_ads_answer(answer: &Value) -> Option<ArxivId> {
218    answer
219        .pointer("/response/docs/0/identifier")?
220        .as_array()?
221        .iter()
222        .filter_map(Value::as_str)
223        .find_map(|id| id.strip_prefix("arXiv:").and_then(arxiv_id))
224}
225
226/// One arXiv search hit.
227#[derive(Debug, Clone, PartialEq, Eq)]
228struct Hit {
229    id: String,
230    title: String,
231    authors: Vec<String>,
232    doi: Option<String>,
233}
234
235/// The first hit in an arXiv search feed that is this paper.
236fn match_search(feed: &str, title: &str, first_author_family: &str, doi: &Doi) -> Option<ArxivId> {
237    let want = normalise(title);
238    let family = first_author_family.to_lowercase();
239    parse_feed(feed).into_iter().find_map(|h| {
240        let same_title = normalise(&h.title) == want;
241        let has_author = h
242            .authors
243            .iter()
244            .any(|a| a.to_lowercase().split_whitespace().any(|w| w == family));
245        let doi_agrees = h
246            .doi
247            .as_deref()
248            .is_none_or(|d| d.eq_ignore_ascii_case(doi.as_str()));
249        (same_title && has_author && doi_agrees)
250            .then(|| h.id.rsplit("/abs/").next().and_then(arxiv_id))
251            .flatten()
252    })
253}
254
255/// Entries of an arXiv Atom feed: id, title, author names, `arxiv:doi`.
256fn parse_feed(feed: &str) -> Vec<Hit> {
257    feed.split("<entry>")
258        .skip(1)
259        .map(|e| {
260            let e = e.split("</entry>").next().unwrap_or(e);
261            Hit {
262                id: tag(e, "id").unwrap_or_default(),
263                title: tag(e, "title").unwrap_or_default(),
264                authors: e
265                    .split("<name>")
266                    .skip(1)
267                    .filter_map(|n| n.split("</name>").next())
268                    .map(unescape)
269                    .collect(),
270                doi: e
271                    .split("<arxiv:doi")
272                    .nth(1)
273                    .and_then(|d| d.split_once('>'))
274                    .and_then(|(_, rest)| rest.split("</arxiv:doi>").next())
275                    .map(unescape),
276            }
277        })
278        .collect()
279}
280
281fn tag(e: &str, name: &str) -> Option<String> {
282    let open = format!("<{name}>");
283    let close = format!("</{name}>");
284    let start = e.find(&open)? + open.len();
285    let end = e[start..].find(&close)? + start;
286    Some(unescape(&e[start..end]))
287}
288
289fn unescape(s: &str) -> String {
290    s.replace("&lt;", "<")
291        .replace("&gt;", ">")
292        .replace("&quot;", "\"")
293        .replace("&#39;", "'")
294        .replace("&amp;", "&")
295        .trim()
296        .to_string()
297}
298
299/// Letters and digits only, case-folded: what two renderings of a title
300/// share once punctuation, markup and line breaks are set aside.
301fn normalise(title: &str) -> String {
302    crate::markup::plain_title(title)
303        .chars()
304        .filter(|c| c.is_alphanumeric())
305        .flat_map(char::to_lowercase)
306        .collect()
307}
308
309/// Whether a title can identify a paper in a search: at least 20 letters
310/// and digits (counted as characters, not bytes -- a CJK title is three
311/// bytes a character) across at least three words. "Introduction" and
312/// "Editorial" are not.
313fn distinctive(title: &str) -> bool {
314    let words = crate::markup::plain_title(title)
315        .split(|c: char| !c.is_alphanumeric())
316        .filter(|w| !w.is_empty())
317        .count();
318    normalise(title).chars().count() >= 20 && words >= 3
319}
320
321/// An arXiv id without its version suffix.
322fn arxiv_id(raw: &str) -> Option<ArxivId> {
323    let raw = raw.trim().trim_end_matches('/');
324    let unversioned = match raw.rsplit_once('v') {
325        Some((head, tail)) if !tail.is_empty() && tail.chars().all(|c| c.is_ascii_digit()) => head,
326        _ => raw,
327    };
328    ArxivId::parse(unversioned).ok()
329}
330
331/// Look for a preprint of `doi`: Crossref's relation, then OpenAlex (when
332/// `openalex_enabled`), then arXiv's search. `crossref_message` is the
333/// record the fetch already holds.
334///
335/// # Errors
336///
337/// A provenance-log failure only; a request that fails is a method that
338/// found nothing, and the next is tried.
339pub async fn find(
340    doi: &Doi,
341    crossref_message: &Value,
342    enabled: &crate::MetadataAccess,
343    ctx: &FetchContext,
344) -> Result<Option<Found>, FetchError> {
345    let openalex_enabled = enabled.openalex;
346    if let Some(arxiv_id) = from_crossref(crossref_message) {
347        return Ok(Some(Found {
348            arxiv_id,
349            found_by: FoundBy::CrossrefRelation,
350        }));
351    }
352    if openalex_enabled {
353        match openalex_work(doi, ctx).await {
354            Ok(Some(work)) => {
355                if let Some(arxiv_id) = from_openalex_work(&work) {
356                    return Ok(Some(Found {
357                        arxiv_id,
358                        found_by: FoundBy::OpenAlexLocation,
359                    }));
360                }
361            }
362            Ok(None) => {}
363            Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
364            Err(e) => tracing::info!(error = %e, "preprint lookup: OpenAlex did not answer"),
365        }
366    }
367    if enabled.inspire {
368        match inspire_record(doi, ctx).await {
369            Ok(Some(record)) => {
370                if let Some(arxiv_id) = from_inspire_record(&record) {
371                    return Ok(Some(Found {
372                        arxiv_id,
373                        found_by: FoundBy::Inspire,
374                    }));
375                }
376            }
377            Ok(None) => {}
378            Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
379            Err(e) => tracing::info!(error = %e, "preprint lookup: INSPIRE did not answer"),
380        }
381    }
382    if enabled.ads {
383        match ads_answer(doi, ctx).await {
384            Ok(Some(answer)) => {
385                if let Some(arxiv_id) = from_ads_answer(&answer) {
386                    return Ok(Some(Found {
387                        arxiv_id,
388                        found_by: FoundBy::Ads,
389                    }));
390                }
391            }
392            Ok(None) => {}
393            Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
394            // A refused token is the user's to fix, so it is said louder than
395            // "did not answer" (#645 review).
396            Err(
397                e @ FetchError::Http(crate::http::HttpError::HttpStatus {
398                    status: 401 | 403, ..
399                }),
400            ) => {
401                tracing::warn!(
402                    error = %e,
403                    "preprint lookup: ADS refused DOIGET_ADS_TOKEN -- check or regenerate the token"
404                );
405            }
406            Err(e) => tracing::info!(error = %e, "preprint lookup: ADS did not answer"),
407        }
408    }
409    let title = crossref_message
410        .pointer("/title/0")
411        .and_then(Value::as_str)
412        .unwrap_or_default();
413    let family = crossref_message
414        .pointer("/author/0/family")
415        .and_then(Value::as_str)
416        .unwrap_or_default();
417    if !distinctive(title) || family.is_empty() {
418        // A short or generic title, or no named first author, is not enough
419        // to tell one paper from another.
420        return Ok(None);
421    }
422    match arxiv_search(doi, title, family, ctx).await {
423        Ok(feed) => Ok(
424            match_search(&feed, title, family, doi).map(|arxiv_id| Found {
425                arxiv_id,
426                found_by: FoundBy::ArxivTitleSearch,
427            }),
428        ),
429        Err(FetchError::Log(e)) => Err(FetchError::Log(e)),
430        Err(e) => {
431            tracing::info!(error = %e, "preprint lookup: arXiv search did not answer");
432            Ok(None)
433        }
434    }
435}
436
437fn base(env: &str, default: &str) -> Result<Url, FetchError> {
438    let raw = std::env::var(env).unwrap_or_else(|_| default.to_string());
439    Url::parse(&raw).map_err(|e| FetchError::SourceSchema {
440        hint: format!("{env}={raw:?} is not a URL: {e}"),
441    })
442}
443
444async fn openalex_work(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
445    let mut url = base("DOIGET_OPENALEX_BASE", "https://api.openalex.org")?;
446    url.set_path(&format!("/works/doi:{}", doi.as_str()));
447    if let Some(email) = crate::orchestrator::configured_contact_email() {
448        url.query_pairs_mut().append_pair("mailto", &email);
449    }
450    let Some(body) = logged_get(doi, "openalex", url, ctx).await? else {
451        return Ok(None);
452    };
453    serde_json::from_slice(&body)
454        .map(Some)
455        .map_err(|e| FetchError::SourceSchema {
456            hint: format!("OpenAlex returned non-JSON: {e}"),
457        })
458}
459
460async fn inspire_record(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
461    let mut url = base("DOIGET_INSPIRE_BASE", "https://inspirehep.net")?;
462    url.set_path(&format!("/api/doi/{}", doi.as_str()));
463    let Some(body) = logged_get(doi, "inspire", url, ctx).await? else {
464        return Ok(None);
465    };
466    serde_json::from_slice(&body)
467        .map(Some)
468        .map_err(|e| FetchError::SourceSchema {
469            hint: format!("INSPIRE returned non-JSON: {e}"),
470        })
471}
472
473async fn ads_answer(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
474    let token = std::env::var(ADS_TOKEN_ENV).unwrap_or_default();
475    if token.trim().is_empty() {
476        return Ok(None);
477    }
478    let mut url = base("DOIGET_ADS_BASE", "https://api.adsabs.harvard.edu")?;
479    url.set_path("/v1/search/query");
480    url.query_pairs_mut()
481        .append_pair("q", &format!("doi:\"{}\"", doi.as_str()))
482        .append_pair("fl", "identifier")
483        .append_pair("rows", "1");
484    let auth = format!("Bearer {}", token.trim());
485    let Some(body) =
486        logged_get_with(doi, "ads", url, &[("Authorization", auth.as_str())], ctx).await?
487    else {
488        return Ok(None);
489    };
490    serde_json::from_slice(&body)
491        .map(Some)
492        .map_err(|e| FetchError::SourceSchema {
493            hint: format!("ADS returned non-JSON: {e}"),
494        })
495}
496
497async fn arxiv_search(
498    doi: &Doi,
499    title: &str,
500    family: &str,
501    ctx: &FetchContext,
502) -> Result<String, FetchError> {
503    let mut url = base("DOIGET_ARXIV_BASE", "https://export.arxiv.org")?;
504    url.set_path("/api/query");
505    // arXiv's query language: a quoted phrase for the title, the surname
506    // for the author. Quotes inside the title would end the phrase early.
507    let phrase: String = title.chars().filter(|c| *c != '"').collect();
508    url.query_pairs_mut()
509        .append_pair("search_query", &format!("ti:\"{phrase}\" AND au:{family}"))
510        .append_pair("max_results", "5");
511    Ok(logged_get(doi, "arxiv", url, ctx)
512        .await?
513        .map(|b| String::from_utf8_lossy(&b).into_owned())
514        .unwrap_or_default())
515}
516
517/// One rate-limited request, recorded in the provenance log as a `resolve`
518/// row for `doi` under `source` -- like every other request a fetch makes
519/// (ADR-0006). `Ok(None)` for a 404.
520async fn logged_get(
521    doi: &Doi,
522    source: &'static str,
523    url: Url,
524    ctx: &FetchContext,
525) -> Result<Option<bytes::Bytes>, FetchError> {
526    logged_get_with(doi, source, url, &[], ctx).await
527}
528
529/// [`logged_get`] with request headers -- values go on the wire only, never
530/// into the provenance row (ADS's token, #644).
531async fn logged_get_with(
532    doi: &Doi,
533    source: &'static str,
534    url: Url,
535    headers: &[(&str, &str)],
536    ctx: &FetchContext,
537) -> Result<Option<bytes::Bytes>, FetchError> {
538    use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
539    let _permit = ctx.rate_limiter.acquire(source).await;
540    let digest = crate::Ref::Doi(doi.clone())
541        .promote(source, None)
542        .digest_hex();
543    let row = |result, size, error_code| RowInput {
544        event: LogEvent::Resolve,
545        result,
546        capability: Capability::Metadata,
547        ref_: Some(doi.as_str()),
548        source: Some(source),
549        error_code,
550        size_bytes: size,
551        license: None,
552        store_path: None,
553        canonical_digest: Some(&digest),
554    };
555    match ctx
556        .http
557        .fetch_bytes_with_headers(source, url, headers)
558        .await
559    {
560        Ok((body, _)) => {
561            ctx.log
562                .append(row(LogResult::Ok, Some(body.len() as u64), None))?;
563            Ok(Some(body))
564        }
565        Err(crate::http::HttpError::HttpStatus { status: 404, .. }) => {
566            ctx.log
567                .append(row(LogResult::Err, None, Some("NOT_FOUND")))?;
568            Ok(None)
569        }
570        Err(e) => {
571            let e = FetchError::Http(e);
572            let code = crate::ErrorCode::from(&e);
573            ctx.log
574                .append(row(LogResult::Err, None, Some(code.as_wire())))?;
575            Err(e)
576        }
577    }
578}
579
580#[cfg(test)]
581#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
582mod tests {
583    use super::*;
584
585    /// Shapes from live Crossref records (2026-09-29): both forms occur.
586    #[test]
587    fn crossref_has_preprint_names_an_arxiv_id_or_an_arxiv_doi() {
588        let by_id = serde_json::json!({"relation": {"has-preprint": [
589            {"id-type": "arxiv", "id": "2302.04668v2", "asserted-by": "subject"}]}});
590        assert_eq!(from_crossref(&by_id).unwrap().as_str(), "2302.04668");
591        let by_doi = serde_json::json!({"relation": {"has-preprint": [
592            {"id-type": "doi", "id": "10.1101/2020.01.01.123456"},
593            {"id-type": "doi", "id": "10.48550/arXiv.2512.07923"}]}});
594        assert_eq!(from_crossref(&by_doi).unwrap().as_str(), "2512.07923");
595        assert!(from_crossref(&serde_json::json!({"relation": {}})).is_none());
596    }
597
598    /// Shape from a live OpenAlex work (10.1038/s41586-019-1666-5).
599    #[test]
600    fn an_openalex_arxiv_location_gives_its_id() {
601        let work = serde_json::json!({"locations": [
602            {"landing_page_url": "https://doi.org/10.1038/s41586-019-1666-5"},
603            {"landing_page_url": "http://arxiv.org/abs/1910.11333", "version": "publishedVersion"}]});
604        assert_eq!(from_openalex_work(&work).unwrap().as_str(), "1910.11333");
605        assert!(from_openalex_work(&serde_json::json!({"locations": []})).is_none());
606    }
607
608    const FEED: &str = r#"<feed><entry>
609<id>http://arxiv.org/abs/2512.07923v1</id>
610<title>Environment-matrix-product operator for boundary-free large-scale
611  quantum many-body simulations</title>
612<author><name>Souta Shimozono</name></author><author><name>Chisa Hotta</name></author>
613</entry><entry>
614<id>http://arxiv.org/abs/2401.00001v2</id>
615<title>Something else entirely</title>
616<author><name>Souta Shimozono</name></author>
617<arxiv:doi xmlns:arxiv="http://arxiv.org/schemas/atom">10.1103/other</arxiv:doi>
618</entry></feed>"#;
619
620    /// Measured on the maintainer's own paper, 2026-09-29.
621    #[test]
622    fn a_search_hit_counts_on_the_same_title_and_first_author() {
623        let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
624        let title = "Environment-matrix-product operator for boundary-free large-scale quantum many-body simulations";
625        assert_eq!(
626            match_search(FEED, title, "Shimozono", &doi)
627                .unwrap()
628                .as_str(),
629            "2512.07923"
630        );
631        assert!(
632            match_search(FEED, title, "Hotta", &doi).is_some(),
633            "any listed author"
634        );
635        assert!(
636            match_search(FEED, title, "White", &doi).is_none(),
637            "wrong author"
638        );
639        assert!(match_search(FEED, "A different title", "Shimozono", &doi).is_none());
640    }
641
642    #[test]
643    fn a_hit_naming_a_different_published_doi_is_not_this_paper() {
644        let feed = FEED.replace("Something else entirely", "Same Title Here Exactly");
645        let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
646        assert!(match_search(&feed, "Same title here, exactly", "Shimozono", &doi).is_none());
647        let other = Doi::parse("10.1103/other").unwrap();
648        assert_eq!(
649            match_search(&feed, "Same title here, exactly", "Shimozono", &other)
650                .unwrap()
651                .as_str(),
652            "2401.00001"
653        );
654    }
655
656    #[test]
657    fn a_generic_or_short_title_is_not_distinctive_counting_characters() {
658        assert!(!distinctive("Introduction"));
659        assert!(!distinctive("Editorial comment"));
660        // Four CJK characters are twelve bytes; they are four characters.
661        assert!(!distinctive("量子多体系"));
662        assert!(distinctive(
663            "Environment-matrix-product operator for boundary-free simulations"
664        ));
665    }
666
667    mod live_shape {
668        //! `find` through the real HTTP client and log, against mocks.
669        use super::super::*;
670        use std::sync::Arc;
671        use wiremock::matchers::{method, path};
672        use wiremock::{Mock, MockServer, ResponseTemplate};
673
674        const TITLE: &str =
675            "Environment-matrix-product operator for boundary-free large-scale quantum many-body simulations";
676
677        async fn ctx(server: &MockServer) -> (FetchContext, tempfile::TempDir) {
678            let host = server.address().to_string();
679            let td = tempfile::TempDir::new().expect("tempdir");
680            let log = camino::Utf8PathBuf::try_from(td.path().join("log.jsonl")).expect("utf-8");
681            std::env::set_var("DOIGET_OPENALEX_BASE", server.uri());
682            std::env::set_var("DOIGET_ARXIV_BASE", server.uri());
683            std::env::set_var("DOIGET_INSPIRE_BASE", server.uri());
684            let sid = "01J0000000000000000000PP62".to_string();
685            (
686                FetchContext {
687                    http: Arc::new(crate::http::HttpClient::new_for_tests_allow_http_multi(&[
688                        ("openalex", host.as_str()),
689                        ("arxiv", host.as_str()),
690                        ("inspire", host.as_str()),
691                    ])),
692                    rate_limiter: Arc::new(crate::rate_limiter::RateLimiter::new(
693                        crate::RateLimits::HARD_CODED,
694                    )),
695                    log: Arc::new(
696                        crate::provenance::ProvenanceLog::open(log, sid.clone()).expect("log"),
697                    ),
698                    session_id: sid,
699                    cache_root: None,
700                },
701                td,
702            )
703        }
704
705        fn clear() {
706            std::env::remove_var("DOIGET_OPENALEX_BASE");
707            std::env::remove_var("DOIGET_ARXIV_BASE");
708            std::env::remove_var("DOIGET_INSPIRE_BASE");
709        }
710
711        async fn server() -> MockServer {
712            let server = MockServer::start().await;
713            Mock::given(method("GET"))
714                .and(path("/works/doi:10.1103/bbnt-brjz"))
715                .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
716                    "locations": [{"landing_page_url": "http://arxiv.org/abs/2512.07923v1"}]
717                })))
718                .mount(&server)
719                .await;
720            Mock::given(method("GET"))
721                .and(path("/api/query"))
722                .respond_with(ResponseTemplate::new(200).set_body_string(format!(
723                    "<feed><entry><id>http://arxiv.org/abs/2512.07923v1</id><title>{TITLE}</title>\
724                     <author><name>Souta Shimozono</name></author></entry></feed>"
725                )))
726                .mount(&server)
727                .await;
728            server
729        }
730
731        fn record(title: &str) -> Value {
732            serde_json::json!({"title": [title], "author": [{"family": "Shimozono"}]})
733        }
734
735        async fn paths(server: &MockServer) -> Vec<String> {
736            server
737                .received_requests()
738                .await
739                .unwrap_or_default()
740                .iter()
741                .map(|r| r.url.path().to_string())
742                .collect()
743        }
744
745        #[tokio::test]
746        #[serial_test::serial]
747        async fn an_enabled_openalex_answers_before_any_arxiv_search() {
748            let server = server().await;
749            let (ctx, _td) = ctx(&server).await;
750            let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
751            let found = find(
752                &doi,
753                &record(TITLE),
754                &crate::MetadataAccess {
755                    openalex: true,
756                    ..Default::default()
757                },
758                &ctx,
759            )
760            .await
761            .unwrap()
762            .unwrap();
763            let seen = paths(&server).await;
764            let log = std::fs::read_to_string(ctx.log.path()).unwrap();
765            clear();
766            assert_eq!(found.found_by, FoundBy::OpenAlexLocation);
767            assert_eq!(found.arxiv_id.as_str(), "2512.07923");
768            assert!(!seen.iter().any(|p| p == "/api/query"), "{seen:?}");
769            assert!(log.contains("\"source\":\"openalex\""), "logged: {log}");
770        }
771
772        /// #642: an enabled INSPIRE answers from `arxiv_eprints`, before
773        /// any arXiv search.
774        #[tokio::test]
775        #[serial_test::serial]
776        async fn an_enabled_inspire_answers_before_the_arxiv_search() {
777            let server = server().await;
778            Mock::given(method("GET"))
779                .and(path("/api/doi/10.1103/bbnt-brjz"))
780                .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
781                    "metadata": {"arxiv_eprints": [{"value": "2512.07923", "categories": ["cond-mat.str-el"]}]}
782                })))
783                .mount(&server)
784                .await;
785            let (ctx, _td) = ctx(&server).await;
786            let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
787            let enabled = crate::MetadataAccess {
788                inspire: true,
789                ..Default::default()
790            };
791            let found = find(&doi, &record(TITLE), &enabled, &ctx)
792                .await
793                .unwrap()
794                .unwrap();
795            let seen = paths(&server).await;
796            clear();
797            assert_eq!(found.found_by, FoundBy::Inspire);
798            assert_eq!(found.arxiv_id.as_str(), "2512.07923");
799            assert!(!seen.iter().any(|p| p == "/api/query"), "{seen:?}");
800            assert!(
801                !seen.iter().any(|p| p.starts_with("/works/")),
802                "OpenAlex off: {seen:?}"
803            );
804        }
805
806        #[tokio::test]
807        #[serial_test::serial]
808        async fn a_disabled_openalex_is_not_asked_and_the_search_answers() {
809            let server = server().await;
810            let (ctx, _td) = ctx(&server).await;
811            let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
812            let found = find(
813                &doi,
814                &record(TITLE),
815                &crate::MetadataAccess::default(),
816                &ctx,
817            )
818            .await
819            .unwrap()
820            .unwrap();
821            let seen = paths(&server).await;
822            let log = std::fs::read_to_string(ctx.log.path()).unwrap();
823            clear();
824            assert_eq!(found.found_by, FoundBy::ArxivTitleSearch);
825            assert!(!seen.iter().any(|p| p.starts_with("/works/")), "{seen:?}");
826            assert!(log.contains("\"source\":\"arxiv\""), "logged: {log}");
827        }
828
829        /// #640: `pubs` is asked only when enabled -- bioRxiv first, then
830        /// medRxiv -- and its preprint DOI is the answer.
831        #[tokio::test]
832        #[serial_test::serial]
833        async fn biorxiv_pubs_is_asked_only_when_enabled_and_names_the_preprint() {
834            let server = MockServer::start().await;
835            Mock::given(method("GET"))
836                .and(path("/pubs/biorxiv/10.1371/journal.pone.0256482/na/json"))
837                .respond_with(ResponseTemplate::new(200).set_body_json(
838                    serde_json::json!({"messages": [{"status": "no posts found"}], "collection": []}),
839                ))
840                .mount(&server)
841                .await;
842            Mock::given(method("GET"))
843                .and(path("/pubs/medrxiv/10.1371/journal.pone.0256482/na/json"))
844                .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
845                    "collection": [{"preprint_doi": "10.1101/2021.04.29.21256344",
846                                    "preprint_platform": "medRxiv"}]
847                })))
848                .mount(&server)
849                .await;
850            let host = server.address().to_string();
851            let td = tempfile::TempDir::new().expect("tempdir");
852            let log = camino::Utf8PathBuf::try_from(td.path().join("log.jsonl")).expect("utf-8");
853            std::env::set_var("DOIGET_BIORXIV_BASE", server.uri());
854            let sid = "01J0000000000000000000BX40".to_string();
855            let ctx = FetchContext {
856                http: Arc::new(crate::http::HttpClient::new_for_tests_allow_http_multi(&[
857                    ("biorxiv", host.as_str()),
858                ])),
859                rate_limiter: Arc::new(crate::rate_limiter::RateLimiter::new(
860                    crate::RateLimits::HARD_CODED,
861                )),
862                log: Arc::new(
863                    crate::provenance::ProvenanceLog::open(log, sid.clone()).expect("log"),
864                ),
865                session_id: sid,
866                cache_root: None,
867            };
868            let doi = Doi::parse("10.1371/journal.pone.0256482").unwrap();
869            let off = find_preprint_doi(&doi, &serde_json::json!({}), false, &ctx)
870                .await
871                .unwrap();
872            let asked_when_off = paths(&server).await;
873            let on = find_preprint_doi(&doi, &serde_json::json!({}), true, &ctx)
874                .await
875                .unwrap()
876                .unwrap();
877            std::env::remove_var("DOIGET_BIORXIV_BASE");
878            assert!(off.is_none());
879            assert!(asked_when_off.is_empty(), "{asked_when_off:?}");
880            assert_eq!(on.doi.as_str(), "10.1101/2021.04.29.21256344");
881            assert_eq!(on.platform.as_deref(), Some("medRxiv"));
882            assert_eq!(on.found_by, FoundBy::BiorxivPubs);
883        }
884
885        #[tokio::test]
886        #[serial_test::serial]
887        async fn a_generic_title_is_not_searched_at_all() {
888            let server = server().await;
889            let (ctx, _td) = ctx(&server).await;
890            let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
891            let found = find(
892                &doi,
893                &record("Introduction"),
894                &crate::MetadataAccess::default(),
895                &ctx,
896            )
897            .await
898            .unwrap();
899            let seen = paths(&server).await;
900            clear();
901            assert!(found.is_none());
902            assert!(seen.is_empty(), "{seen:?}");
903        }
904    }
905
906    /// #644: an ADS search answer's `identifier` list; the `arXiv:` entry
907    /// is the id, the bibcode and DOI are not.
908    #[test]
909    fn an_ads_answer_gives_its_arxiv_identifier() {
910        let answer = serde_json::json!({"response": {"docs": [{"identifier": [
911            "2016PhRvL.116f1102A", "10.1103/PhysRevLett.116.061102", "arXiv:1602.03837"]}]}});
912        assert_eq!(from_ads_answer(&answer).unwrap().as_str(), "1602.03837");
913        let none =
914            serde_json::json!({"response": {"docs": [{"identifier": ["2016PhRvL.116f1102A"]}]}});
915        assert!(from_ads_answer(&none).is_none());
916        assert!(from_ads_answer(&serde_json::json!({"response": {"docs": []}})).is_none());
917    }
918
919    /// #642: the shape of a live INSPIRE record (10.1103/PhysRevLett.116.061102).
920    #[test]
921    fn an_inspire_record_gives_its_arxiv_eprint() {
922        let rec = serde_json::json!({"metadata": {
923            "arxiv_eprints": [{"value": "1602.03837", "categories": ["gr-qc"]}],
924            "documents": [{"url": "https://inspirehep.net/files/4d19c13c"}]}});
925        assert_eq!(from_inspire_record(&rec).unwrap().as_str(), "1602.03837");
926        assert!(from_inspire_record(&serde_json::json!({"metadata": {}})).is_none());
927    }
928
929    /// #640: shapes from a live Crossref sample and a live `pubs` answer.
930    #[test]
931    fn a_non_arxiv_preprint_doi_is_read_from_crossref_or_pubs() {
932        let rel = serde_json::json!({"relation": {"has-preprint": [
933            {"id-type": "doi", "id": "10.48550/arXiv.2512.07923"},
934            {"id-type": "doi", "id": "10.1101/2021.04.29.21256344"}]}});
935        assert_eq!(
936            preprint_doi_from_crossref(&rel).unwrap().as_str(),
937            "10.1101/2021.04.29.21256344",
938            "the arXiv DOI is from_crossref's; this is the other one"
939        );
940        let arxiv_only = serde_json::json!({"relation": {"has-preprint": [
941            {"id-type": "arxiv", "id": "2302.04668v2"}]}});
942        assert!(preprint_doi_from_crossref(&arxiv_only).is_none());
943        let pubs = serde_json::json!({"messages": [{"status": "ok"}], "collection": [{
944            "preprint_doi": "10.1101/2021.04.29.21256344",
945            "published_doi": "10.1371/journal.pone.0256482",
946            "preprint_platform": "medRxiv"}]});
947        let (found, platform) = from_pubs(&pubs).unwrap();
948        assert_eq!(found.as_str(), "10.1101/2021.04.29.21256344");
949        assert_eq!(platform.as_deref(), Some("medRxiv"));
950        assert!(from_pubs(&serde_json::json!({"collection": []})).is_none());
951    }
952}