1use serde_json::Value;
26use url::Url;
27
28use crate::source::{FetchContext, FetchError};
29
30pub const ADS_TOKEN_ENV: &str = "DOIGET_ADS_TOKEN";
32use crate::{ArxivId, Doi};
33
34#[derive(Debug, Clone, Copy, PartialEq, Eq)]
36pub enum FoundBy {
37 Unpaywall,
39 CrossrefRelation,
41 OpenAlexLocation,
43 ArxivTitleSearch,
45 BiorxivPubs,
47 Inspire,
49 Ads,
51}
52
53impl FoundBy {
54 #[must_use]
56 pub const fn as_str(self) -> &'static str {
57 match self {
58 Self::Unpaywall => "unpaywall",
59 Self::CrossrefRelation => "crossref_relation",
60 Self::OpenAlexLocation => "openalex_location",
61 Self::ArxivTitleSearch => "arxiv_title_search",
62 Self::BiorxivPubs => "biorxiv_pubs",
63 Self::Inspire => "inspire",
64 Self::Ads => "ads",
65 }
66 }
67}
68
69#[derive(Debug, Clone, PartialEq, Eq)]
71pub struct Found {
72 pub arxiv_id: ArxivId,
74 pub found_by: FoundBy,
76}
77
78#[must_use]
80pub fn from_crossref(crossref_message: &Value) -> Option<ArxivId> {
81 let rels = crossref_message
82 .pointer("/relation/has-preprint")?
83 .as_array()?;
84 rels.iter().find_map(|r| {
85 let id = r.get("id").and_then(Value::as_str)?.trim();
86 match r.get("id-type").and_then(Value::as_str)? {
87 "arxiv" => arxiv_id(id),
88 "doi" => {
89 let lower = id.to_ascii_lowercase();
90 lower
91 .strip_prefix("10.48550/arxiv.")
92 .and_then(|_| arxiv_id(&id["10.48550/arxiv.".len()..]))
93 }
94 _ => None,
95 }
96 })
97}
98
99#[derive(Debug, Clone, PartialEq, Eq)]
102pub struct FoundDoi {
103 pub doi: Doi,
105 pub platform: Option<String>,
107 pub found_by: FoundBy,
109}
110
111#[must_use]
114pub fn preprint_doi_from_crossref(crossref_message: &Value) -> Option<Doi> {
115 let rels = crossref_message
116 .pointer("/relation/has-preprint")?
117 .as_array()?;
118 rels.iter().find_map(|r| {
119 (r.get("id-type").and_then(Value::as_str)? == "doi")
120 .then(|| r.get("id").and_then(Value::as_str))
121 .flatten()
122 .map(str::trim)
123 .filter(|id| !id.to_ascii_lowercase().starts_with("10.48550/"))
124 .and_then(|id| Doi::parse(id).ok())
125 })
126}
127
128fn from_pubs(answer: &Value) -> Option<(Doi, Option<String>)> {
130 let first = answer.get("collection")?.as_array()?.first()?;
131 let doi = Doi::parse(first.get("preprint_doi")?.as_str()?.trim()).ok()?;
132 let platform = first
133 .get("preprint_platform")
134 .and_then(Value::as_str)
135 .map(str::to_string);
136 Some((doi, platform))
137}
138
139pub async fn find_preprint_doi(
147 doi: &Doi,
148 crossref_message: &Value,
149 biorxiv_enabled: bool,
150 ctx: &FetchContext,
151) -> Result<Option<FoundDoi>, FetchError> {
152 if let Some(found) = preprint_doi_from_crossref(crossref_message) {
153 return Ok(Some(FoundDoi {
154 doi: found,
155 platform: None,
156 found_by: FoundBy::CrossrefRelation,
157 }));
158 }
159 if !biorxiv_enabled {
160 return Ok(None);
161 }
162 for server in ["biorxiv", "medrxiv"] {
163 let mut url = base("DOIGET_BIORXIV_BASE", "https://api.biorxiv.org")?;
164 url.set_path(&format!("/pubs/{server}/{}/na/json", doi.as_str()));
165 match logged_get(doi, "biorxiv", url, ctx).await {
166 Ok(Some(body)) => {
167 if let Some((found, platform)) = serde_json::from_slice::<Value>(&body)
168 .ok()
169 .as_ref()
170 .and_then(from_pubs)
171 {
172 return Ok(Some(FoundDoi {
173 doi: found,
174 platform,
175 found_by: FoundBy::BiorxivPubs,
176 }));
177 }
178 }
179 Ok(None) => {}
180 Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
181 Err(e) => {
182 tracing::info!(error = %e, server, "preprint lookup: bioRxiv pubs did not answer")
183 }
184 }
185 }
186 Ok(None)
187}
188
189#[must_use]
191pub fn from_openalex_work(work: &Value) -> Option<ArxivId> {
192 work.get("locations")?.as_array()?.iter().find_map(|l| {
193 let url = l.get("landing_page_url").and_then(Value::as_str)?;
194 let rest = url
195 .strip_prefix("http://arxiv.org/abs/")
196 .or_else(|| url.strip_prefix("https://arxiv.org/abs/"))?;
197 arxiv_id(rest)
198 })
199}
200
201#[must_use]
205pub fn from_inspire_record(record: &Value) -> Option<ArxivId> {
206 record
207 .pointer("/metadata/arxiv_eprints")?
208 .as_array()?
209 .iter()
210 .find_map(|e| e.get("value").and_then(Value::as_str).and_then(arxiv_id))
211}
212
213#[must_use]
217pub fn from_ads_answer(answer: &Value) -> Option<ArxivId> {
218 answer
219 .pointer("/response/docs/0/identifier")?
220 .as_array()?
221 .iter()
222 .filter_map(Value::as_str)
223 .find_map(|id| id.strip_prefix("arXiv:").and_then(arxiv_id))
224}
225
226#[derive(Debug, Clone, PartialEq, Eq)]
228struct Hit {
229 id: String,
230 title: String,
231 authors: Vec<String>,
232 doi: Option<String>,
233}
234
235fn match_search(feed: &str, title: &str, first_author_family: &str, doi: &Doi) -> Option<ArxivId> {
237 let want = normalise(title);
238 let family = first_author_family.to_lowercase();
239 parse_feed(feed).into_iter().find_map(|h| {
240 let same_title = normalise(&h.title) == want;
241 let has_author = h
242 .authors
243 .iter()
244 .any(|a| a.to_lowercase().split_whitespace().any(|w| w == family));
245 let doi_agrees = h
246 .doi
247 .as_deref()
248 .is_none_or(|d| d.eq_ignore_ascii_case(doi.as_str()));
249 (same_title && has_author && doi_agrees)
250 .then(|| h.id.rsplit("/abs/").next().and_then(arxiv_id))
251 .flatten()
252 })
253}
254
255fn parse_feed(feed: &str) -> Vec<Hit> {
257 feed.split("<entry>")
258 .skip(1)
259 .map(|e| {
260 let e = e.split("</entry>").next().unwrap_or(e);
261 Hit {
262 id: tag(e, "id").unwrap_or_default(),
263 title: tag(e, "title").unwrap_or_default(),
264 authors: e
265 .split("<name>")
266 .skip(1)
267 .filter_map(|n| n.split("</name>").next())
268 .map(unescape)
269 .collect(),
270 doi: e
271 .split("<arxiv:doi")
272 .nth(1)
273 .and_then(|d| d.split_once('>'))
274 .and_then(|(_, rest)| rest.split("</arxiv:doi>").next())
275 .map(unescape),
276 }
277 })
278 .collect()
279}
280
281fn tag(e: &str, name: &str) -> Option<String> {
282 let open = format!("<{name}>");
283 let close = format!("</{name}>");
284 let start = e.find(&open)? + open.len();
285 let end = e[start..].find(&close)? + start;
286 Some(unescape(&e[start..end]))
287}
288
289fn unescape(s: &str) -> String {
290 s.replace("<", "<")
291 .replace(">", ">")
292 .replace(""", "\"")
293 .replace("'", "'")
294 .replace("&", "&")
295 .trim()
296 .to_string()
297}
298
299fn normalise(title: &str) -> String {
302 crate::markup::plain_title(title)
303 .chars()
304 .filter(|c| c.is_alphanumeric())
305 .flat_map(char::to_lowercase)
306 .collect()
307}
308
309fn distinctive(title: &str) -> bool {
314 let words = crate::markup::plain_title(title)
315 .split(|c: char| !c.is_alphanumeric())
316 .filter(|w| !w.is_empty())
317 .count();
318 normalise(title).chars().count() >= 20 && words >= 3
319}
320
321fn arxiv_id(raw: &str) -> Option<ArxivId> {
323 let raw = raw.trim().trim_end_matches('/');
324 let unversioned = match raw.rsplit_once('v') {
325 Some((head, tail)) if !tail.is_empty() && tail.chars().all(|c| c.is_ascii_digit()) => head,
326 _ => raw,
327 };
328 ArxivId::parse(unversioned).ok()
329}
330
331pub async fn find(
340 doi: &Doi,
341 crossref_message: &Value,
342 enabled: &crate::MetadataAccess,
343 ctx: &FetchContext,
344) -> Result<Option<Found>, FetchError> {
345 let openalex_enabled = enabled.openalex;
346 if let Some(arxiv_id) = from_crossref(crossref_message) {
347 return Ok(Some(Found {
348 arxiv_id,
349 found_by: FoundBy::CrossrefRelation,
350 }));
351 }
352 if openalex_enabled {
353 match openalex_work(doi, ctx).await {
354 Ok(Some(work)) => {
355 if let Some(arxiv_id) = from_openalex_work(&work) {
356 return Ok(Some(Found {
357 arxiv_id,
358 found_by: FoundBy::OpenAlexLocation,
359 }));
360 }
361 }
362 Ok(None) => {}
363 Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
364 Err(e) => tracing::info!(error = %e, "preprint lookup: OpenAlex did not answer"),
365 }
366 }
367 if enabled.inspire {
368 match inspire_record(doi, ctx).await {
369 Ok(Some(record)) => {
370 if let Some(arxiv_id) = from_inspire_record(&record) {
371 return Ok(Some(Found {
372 arxiv_id,
373 found_by: FoundBy::Inspire,
374 }));
375 }
376 }
377 Ok(None) => {}
378 Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
379 Err(e) => tracing::info!(error = %e, "preprint lookup: INSPIRE did not answer"),
380 }
381 }
382 if enabled.ads {
383 match ads_answer(doi, ctx).await {
384 Ok(Some(answer)) => {
385 if let Some(arxiv_id) = from_ads_answer(&answer) {
386 return Ok(Some(Found {
387 arxiv_id,
388 found_by: FoundBy::Ads,
389 }));
390 }
391 }
392 Ok(None) => {}
393 Err(FetchError::Log(e)) => return Err(FetchError::Log(e)),
394 Err(
397 e @ FetchError::Http(crate::http::HttpError::HttpStatus {
398 status: 401 | 403, ..
399 }),
400 ) => {
401 tracing::warn!(
402 error = %e,
403 "preprint lookup: ADS refused DOIGET_ADS_TOKEN -- check or regenerate the token"
404 );
405 }
406 Err(e) => tracing::info!(error = %e, "preprint lookup: ADS did not answer"),
407 }
408 }
409 let title = crossref_message
410 .pointer("/title/0")
411 .and_then(Value::as_str)
412 .unwrap_or_default();
413 let family = crossref_message
414 .pointer("/author/0/family")
415 .and_then(Value::as_str)
416 .unwrap_or_default();
417 if !distinctive(title) || family.is_empty() {
418 return Ok(None);
421 }
422 match arxiv_search(doi, title, family, ctx).await {
423 Ok(feed) => Ok(
424 match_search(&feed, title, family, doi).map(|arxiv_id| Found {
425 arxiv_id,
426 found_by: FoundBy::ArxivTitleSearch,
427 }),
428 ),
429 Err(FetchError::Log(e)) => Err(FetchError::Log(e)),
430 Err(e) => {
431 tracing::info!(error = %e, "preprint lookup: arXiv search did not answer");
432 Ok(None)
433 }
434 }
435}
436
437fn base(env: &str, default: &str) -> Result<Url, FetchError> {
438 let raw = std::env::var(env).unwrap_or_else(|_| default.to_string());
439 Url::parse(&raw).map_err(|e| FetchError::SourceSchema {
440 hint: format!("{env}={raw:?} is not a URL: {e}"),
441 })
442}
443
444async fn openalex_work(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
445 let mut url = base("DOIGET_OPENALEX_BASE", "https://api.openalex.org")?;
446 url.set_path(&format!("/works/doi:{}", doi.as_str()));
447 if let Some(email) = crate::orchestrator::configured_contact_email() {
448 url.query_pairs_mut().append_pair("mailto", &email);
449 }
450 let Some(body) = logged_get(doi, "openalex", url, ctx).await? else {
451 return Ok(None);
452 };
453 serde_json::from_slice(&body)
454 .map(Some)
455 .map_err(|e| FetchError::SourceSchema {
456 hint: format!("OpenAlex returned non-JSON: {e}"),
457 })
458}
459
460async fn inspire_record(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
461 let mut url = base("DOIGET_INSPIRE_BASE", "https://inspirehep.net")?;
462 url.set_path(&format!("/api/doi/{}", doi.as_str()));
463 let Some(body) = logged_get(doi, "inspire", url, ctx).await? else {
464 return Ok(None);
465 };
466 serde_json::from_slice(&body)
467 .map(Some)
468 .map_err(|e| FetchError::SourceSchema {
469 hint: format!("INSPIRE returned non-JSON: {e}"),
470 })
471}
472
473async fn ads_answer(doi: &Doi, ctx: &FetchContext) -> Result<Option<Value>, FetchError> {
474 let token = std::env::var(ADS_TOKEN_ENV).unwrap_or_default();
475 if token.trim().is_empty() {
476 return Ok(None);
477 }
478 let mut url = base("DOIGET_ADS_BASE", "https://api.adsabs.harvard.edu")?;
479 url.set_path("/v1/search/query");
480 url.query_pairs_mut()
481 .append_pair("q", &format!("doi:\"{}\"", doi.as_str()))
482 .append_pair("fl", "identifier")
483 .append_pair("rows", "1");
484 let auth = format!("Bearer {}", token.trim());
485 let Some(body) =
486 logged_get_with(doi, "ads", url, &[("Authorization", auth.as_str())], ctx).await?
487 else {
488 return Ok(None);
489 };
490 serde_json::from_slice(&body)
491 .map(Some)
492 .map_err(|e| FetchError::SourceSchema {
493 hint: format!("ADS returned non-JSON: {e}"),
494 })
495}
496
497async fn arxiv_search(
498 doi: &Doi,
499 title: &str,
500 family: &str,
501 ctx: &FetchContext,
502) -> Result<String, FetchError> {
503 let mut url = base("DOIGET_ARXIV_BASE", "https://export.arxiv.org")?;
504 url.set_path("/api/query");
505 let phrase: String = title.chars().filter(|c| *c != '"').collect();
508 url.query_pairs_mut()
509 .append_pair("search_query", &format!("ti:\"{phrase}\" AND au:{family}"))
510 .append_pair("max_results", "5");
511 Ok(logged_get(doi, "arxiv", url, ctx)
512 .await?
513 .map(|b| String::from_utf8_lossy(&b).into_owned())
514 .unwrap_or_default())
515}
516
517async fn logged_get(
521 doi: &Doi,
522 source: &'static str,
523 url: Url,
524 ctx: &FetchContext,
525) -> Result<Option<bytes::Bytes>, FetchError> {
526 logged_get_with(doi, source, url, &[], ctx).await
527}
528
529async fn logged_get_with(
532 doi: &Doi,
533 source: &'static str,
534 url: Url,
535 headers: &[(&str, &str)],
536 ctx: &FetchContext,
537) -> Result<Option<bytes::Bytes>, FetchError> {
538 use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
539 let _permit = ctx.rate_limiter.acquire(source).await;
540 let digest = crate::Ref::Doi(doi.clone())
541 .promote(source, None)
542 .digest_hex();
543 let row = |result, size, error_code| RowInput {
544 event: LogEvent::Resolve,
545 result,
546 capability: Capability::Metadata,
547 ref_: Some(doi.as_str()),
548 source: Some(source),
549 error_code,
550 size_bytes: size,
551 license: None,
552 store_path: None,
553 canonical_digest: Some(&digest),
554 };
555 match ctx
556 .http
557 .fetch_bytes_with_headers(source, url, headers)
558 .await
559 {
560 Ok((body, _)) => {
561 ctx.log
562 .append(row(LogResult::Ok, Some(body.len() as u64), None))?;
563 Ok(Some(body))
564 }
565 Err(crate::http::HttpError::HttpStatus { status: 404, .. }) => {
566 ctx.log
567 .append(row(LogResult::Err, None, Some("NOT_FOUND")))?;
568 Ok(None)
569 }
570 Err(e) => {
571 let e = FetchError::Http(e);
572 let code = crate::ErrorCode::from(&e);
573 ctx.log
574 .append(row(LogResult::Err, None, Some(code.as_wire())))?;
575 Err(e)
576 }
577 }
578}
579
580#[cfg(test)]
581#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
582mod tests {
583 use super::*;
584
585 #[test]
587 fn crossref_has_preprint_names_an_arxiv_id_or_an_arxiv_doi() {
588 let by_id = serde_json::json!({"relation": {"has-preprint": [
589 {"id-type": "arxiv", "id": "2302.04668v2", "asserted-by": "subject"}]}});
590 assert_eq!(from_crossref(&by_id).unwrap().as_str(), "2302.04668");
591 let by_doi = serde_json::json!({"relation": {"has-preprint": [
592 {"id-type": "doi", "id": "10.1101/2020.01.01.123456"},
593 {"id-type": "doi", "id": "10.48550/arXiv.2512.07923"}]}});
594 assert_eq!(from_crossref(&by_doi).unwrap().as_str(), "2512.07923");
595 assert!(from_crossref(&serde_json::json!({"relation": {}})).is_none());
596 }
597
598 #[test]
600 fn an_openalex_arxiv_location_gives_its_id() {
601 let work = serde_json::json!({"locations": [
602 {"landing_page_url": "https://doi.org/10.1038/s41586-019-1666-5"},
603 {"landing_page_url": "http://arxiv.org/abs/1910.11333", "version": "publishedVersion"}]});
604 assert_eq!(from_openalex_work(&work).unwrap().as_str(), "1910.11333");
605 assert!(from_openalex_work(&serde_json::json!({"locations": []})).is_none());
606 }
607
608 const FEED: &str = r#"<feed><entry>
609<id>http://arxiv.org/abs/2512.07923v1</id>
610<title>Environment-matrix-product operator for boundary-free large-scale
611 quantum many-body simulations</title>
612<author><name>Souta Shimozono</name></author><author><name>Chisa Hotta</name></author>
613</entry><entry>
614<id>http://arxiv.org/abs/2401.00001v2</id>
615<title>Something else entirely</title>
616<author><name>Souta Shimozono</name></author>
617<arxiv:doi xmlns:arxiv="http://arxiv.org/schemas/atom">10.1103/other</arxiv:doi>
618</entry></feed>"#;
619
620 #[test]
622 fn a_search_hit_counts_on_the_same_title_and_first_author() {
623 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
624 let title = "Environment-matrix-product operator for boundary-free large-scale quantum many-body simulations";
625 assert_eq!(
626 match_search(FEED, title, "Shimozono", &doi)
627 .unwrap()
628 .as_str(),
629 "2512.07923"
630 );
631 assert!(
632 match_search(FEED, title, "Hotta", &doi).is_some(),
633 "any listed author"
634 );
635 assert!(
636 match_search(FEED, title, "White", &doi).is_none(),
637 "wrong author"
638 );
639 assert!(match_search(FEED, "A different title", "Shimozono", &doi).is_none());
640 }
641
642 #[test]
643 fn a_hit_naming_a_different_published_doi_is_not_this_paper() {
644 let feed = FEED.replace("Something else entirely", "Same Title Here Exactly");
645 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
646 assert!(match_search(&feed, "Same title here, exactly", "Shimozono", &doi).is_none());
647 let other = Doi::parse("10.1103/other").unwrap();
648 assert_eq!(
649 match_search(&feed, "Same title here, exactly", "Shimozono", &other)
650 .unwrap()
651 .as_str(),
652 "2401.00001"
653 );
654 }
655
656 #[test]
657 fn a_generic_or_short_title_is_not_distinctive_counting_characters() {
658 assert!(!distinctive("Introduction"));
659 assert!(!distinctive("Editorial comment"));
660 assert!(!distinctive("量子多体系"));
662 assert!(distinctive(
663 "Environment-matrix-product operator for boundary-free simulations"
664 ));
665 }
666
667 mod live_shape {
668 use super::super::*;
670 use std::sync::Arc;
671 use wiremock::matchers::{method, path};
672 use wiremock::{Mock, MockServer, ResponseTemplate};
673
674 const TITLE: &str =
675 "Environment-matrix-product operator for boundary-free large-scale quantum many-body simulations";
676
677 async fn ctx(server: &MockServer) -> (FetchContext, tempfile::TempDir) {
678 let host = server.address().to_string();
679 let td = tempfile::TempDir::new().expect("tempdir");
680 let log = camino::Utf8PathBuf::try_from(td.path().join("log.jsonl")).expect("utf-8");
681 std::env::set_var("DOIGET_OPENALEX_BASE", server.uri());
682 std::env::set_var("DOIGET_ARXIV_BASE", server.uri());
683 std::env::set_var("DOIGET_INSPIRE_BASE", server.uri());
684 let sid = "01J0000000000000000000PP62".to_string();
685 (
686 FetchContext {
687 http: Arc::new(crate::http::HttpClient::new_for_tests_allow_http_multi(&[
688 ("openalex", host.as_str()),
689 ("arxiv", host.as_str()),
690 ("inspire", host.as_str()),
691 ])),
692 rate_limiter: Arc::new(crate::rate_limiter::RateLimiter::new(
693 crate::RateLimits::HARD_CODED,
694 )),
695 log: Arc::new(
696 crate::provenance::ProvenanceLog::open(log, sid.clone()).expect("log"),
697 ),
698 session_id: sid,
699 cache_root: None,
700 },
701 td,
702 )
703 }
704
705 fn clear() {
706 std::env::remove_var("DOIGET_OPENALEX_BASE");
707 std::env::remove_var("DOIGET_ARXIV_BASE");
708 std::env::remove_var("DOIGET_INSPIRE_BASE");
709 }
710
711 async fn server() -> MockServer {
712 let server = MockServer::start().await;
713 Mock::given(method("GET"))
714 .and(path("/works/doi:10.1103/bbnt-brjz"))
715 .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
716 "locations": [{"landing_page_url": "http://arxiv.org/abs/2512.07923v1"}]
717 })))
718 .mount(&server)
719 .await;
720 Mock::given(method("GET"))
721 .and(path("/api/query"))
722 .respond_with(ResponseTemplate::new(200).set_body_string(format!(
723 "<feed><entry><id>http://arxiv.org/abs/2512.07923v1</id><title>{TITLE}</title>\
724 <author><name>Souta Shimozono</name></author></entry></feed>"
725 )))
726 .mount(&server)
727 .await;
728 server
729 }
730
731 fn record(title: &str) -> Value {
732 serde_json::json!({"title": [title], "author": [{"family": "Shimozono"}]})
733 }
734
735 async fn paths(server: &MockServer) -> Vec<String> {
736 server
737 .received_requests()
738 .await
739 .unwrap_or_default()
740 .iter()
741 .map(|r| r.url.path().to_string())
742 .collect()
743 }
744
745 #[tokio::test]
746 #[serial_test::serial]
747 async fn an_enabled_openalex_answers_before_any_arxiv_search() {
748 let server = server().await;
749 let (ctx, _td) = ctx(&server).await;
750 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
751 let found = find(
752 &doi,
753 &record(TITLE),
754 &crate::MetadataAccess {
755 openalex: true,
756 ..Default::default()
757 },
758 &ctx,
759 )
760 .await
761 .unwrap()
762 .unwrap();
763 let seen = paths(&server).await;
764 let log = std::fs::read_to_string(ctx.log.path()).unwrap();
765 clear();
766 assert_eq!(found.found_by, FoundBy::OpenAlexLocation);
767 assert_eq!(found.arxiv_id.as_str(), "2512.07923");
768 assert!(!seen.iter().any(|p| p == "/api/query"), "{seen:?}");
769 assert!(log.contains("\"source\":\"openalex\""), "logged: {log}");
770 }
771
772 #[tokio::test]
775 #[serial_test::serial]
776 async fn an_enabled_inspire_answers_before_the_arxiv_search() {
777 let server = server().await;
778 Mock::given(method("GET"))
779 .and(path("/api/doi/10.1103/bbnt-brjz"))
780 .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
781 "metadata": {"arxiv_eprints": [{"value": "2512.07923", "categories": ["cond-mat.str-el"]}]}
782 })))
783 .mount(&server)
784 .await;
785 let (ctx, _td) = ctx(&server).await;
786 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
787 let enabled = crate::MetadataAccess {
788 inspire: true,
789 ..Default::default()
790 };
791 let found = find(&doi, &record(TITLE), &enabled, &ctx)
792 .await
793 .unwrap()
794 .unwrap();
795 let seen = paths(&server).await;
796 clear();
797 assert_eq!(found.found_by, FoundBy::Inspire);
798 assert_eq!(found.arxiv_id.as_str(), "2512.07923");
799 assert!(!seen.iter().any(|p| p == "/api/query"), "{seen:?}");
800 assert!(
801 !seen.iter().any(|p| p.starts_with("/works/")),
802 "OpenAlex off: {seen:?}"
803 );
804 }
805
806 #[tokio::test]
807 #[serial_test::serial]
808 async fn a_disabled_openalex_is_not_asked_and_the_search_answers() {
809 let server = server().await;
810 let (ctx, _td) = ctx(&server).await;
811 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
812 let found = find(
813 &doi,
814 &record(TITLE),
815 &crate::MetadataAccess::default(),
816 &ctx,
817 )
818 .await
819 .unwrap()
820 .unwrap();
821 let seen = paths(&server).await;
822 let log = std::fs::read_to_string(ctx.log.path()).unwrap();
823 clear();
824 assert_eq!(found.found_by, FoundBy::ArxivTitleSearch);
825 assert!(!seen.iter().any(|p| p.starts_with("/works/")), "{seen:?}");
826 assert!(log.contains("\"source\":\"arxiv\""), "logged: {log}");
827 }
828
829 #[tokio::test]
832 #[serial_test::serial]
833 async fn biorxiv_pubs_is_asked_only_when_enabled_and_names_the_preprint() {
834 let server = MockServer::start().await;
835 Mock::given(method("GET"))
836 .and(path("/pubs/biorxiv/10.1371/journal.pone.0256482/na/json"))
837 .respond_with(ResponseTemplate::new(200).set_body_json(
838 serde_json::json!({"messages": [{"status": "no posts found"}], "collection": []}),
839 ))
840 .mount(&server)
841 .await;
842 Mock::given(method("GET"))
843 .and(path("/pubs/medrxiv/10.1371/journal.pone.0256482/na/json"))
844 .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({
845 "collection": [{"preprint_doi": "10.1101/2021.04.29.21256344",
846 "preprint_platform": "medRxiv"}]
847 })))
848 .mount(&server)
849 .await;
850 let host = server.address().to_string();
851 let td = tempfile::TempDir::new().expect("tempdir");
852 let log = camino::Utf8PathBuf::try_from(td.path().join("log.jsonl")).expect("utf-8");
853 std::env::set_var("DOIGET_BIORXIV_BASE", server.uri());
854 let sid = "01J0000000000000000000BX40".to_string();
855 let ctx = FetchContext {
856 http: Arc::new(crate::http::HttpClient::new_for_tests_allow_http_multi(&[
857 ("biorxiv", host.as_str()),
858 ])),
859 rate_limiter: Arc::new(crate::rate_limiter::RateLimiter::new(
860 crate::RateLimits::HARD_CODED,
861 )),
862 log: Arc::new(
863 crate::provenance::ProvenanceLog::open(log, sid.clone()).expect("log"),
864 ),
865 session_id: sid,
866 cache_root: None,
867 };
868 let doi = Doi::parse("10.1371/journal.pone.0256482").unwrap();
869 let off = find_preprint_doi(&doi, &serde_json::json!({}), false, &ctx)
870 .await
871 .unwrap();
872 let asked_when_off = paths(&server).await;
873 let on = find_preprint_doi(&doi, &serde_json::json!({}), true, &ctx)
874 .await
875 .unwrap()
876 .unwrap();
877 std::env::remove_var("DOIGET_BIORXIV_BASE");
878 assert!(off.is_none());
879 assert!(asked_when_off.is_empty(), "{asked_when_off:?}");
880 assert_eq!(on.doi.as_str(), "10.1101/2021.04.29.21256344");
881 assert_eq!(on.platform.as_deref(), Some("medRxiv"));
882 assert_eq!(on.found_by, FoundBy::BiorxivPubs);
883 }
884
885 #[tokio::test]
886 #[serial_test::serial]
887 async fn a_generic_title_is_not_searched_at_all() {
888 let server = server().await;
889 let (ctx, _td) = ctx(&server).await;
890 let doi = Doi::parse("10.1103/bbnt-brjz").unwrap();
891 let found = find(
892 &doi,
893 &record("Introduction"),
894 &crate::MetadataAccess::default(),
895 &ctx,
896 )
897 .await
898 .unwrap();
899 let seen = paths(&server).await;
900 clear();
901 assert!(found.is_none());
902 assert!(seen.is_empty(), "{seen:?}");
903 }
904 }
905
906 #[test]
909 fn an_ads_answer_gives_its_arxiv_identifier() {
910 let answer = serde_json::json!({"response": {"docs": [{"identifier": [
911 "2016PhRvL.116f1102A", "10.1103/PhysRevLett.116.061102", "arXiv:1602.03837"]}]}});
912 assert_eq!(from_ads_answer(&answer).unwrap().as_str(), "1602.03837");
913 let none =
914 serde_json::json!({"response": {"docs": [{"identifier": ["2016PhRvL.116f1102A"]}]}});
915 assert!(from_ads_answer(&none).is_none());
916 assert!(from_ads_answer(&serde_json::json!({"response": {"docs": []}})).is_none());
917 }
918
919 #[test]
921 fn an_inspire_record_gives_its_arxiv_eprint() {
922 let rec = serde_json::json!({"metadata": {
923 "arxiv_eprints": [{"value": "1602.03837", "categories": ["gr-qc"]}],
924 "documents": [{"url": "https://inspirehep.net/files/4d19c13c"}]}});
925 assert_eq!(from_inspire_record(&rec).unwrap().as_str(), "1602.03837");
926 assert!(from_inspire_record(&serde_json::json!({"metadata": {}})).is_none());
927 }
928
929 #[test]
931 fn a_non_arxiv_preprint_doi_is_read_from_crossref_or_pubs() {
932 let rel = serde_json::json!({"relation": {"has-preprint": [
933 {"id-type": "doi", "id": "10.48550/arXiv.2512.07923"},
934 {"id-type": "doi", "id": "10.1101/2021.04.29.21256344"}]}});
935 assert_eq!(
936 preprint_doi_from_crossref(&rel).unwrap().as_str(),
937 "10.1101/2021.04.29.21256344",
938 "the arXiv DOI is from_crossref's; this is the other one"
939 );
940 let arxiv_only = serde_json::json!({"relation": {"has-preprint": [
941 {"id-type": "arxiv", "id": "2302.04668v2"}]}});
942 assert!(preprint_doi_from_crossref(&arxiv_only).is_none());
943 let pubs = serde_json::json!({"messages": [{"status": "ok"}], "collection": [{
944 "preprint_doi": "10.1101/2021.04.29.21256344",
945 "published_doi": "10.1371/journal.pone.0256482",
946 "preprint_platform": "medRxiv"}]});
947 let (found, platform) = from_pubs(&pubs).unwrap();
948 assert_eq!(found.as_str(), "10.1101/2021.04.29.21256344");
949 assert_eq!(platform.as_deref(), Some("medRxiv"));
950 assert!(from_pubs(&serde_json::json!({"collection": []})).is_none());
951 }
952}