1use camino::{Utf8Path, Utf8PathBuf};
48use chrono::{DateTime, Duration, Utc};
49use quick_xml::events::Event;
50use quick_xml::Reader;
51use serde::{Deserialize, Serialize};
52use url::Url;
53
54use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
55use crate::source::{FetchContext, FetchError};
56use crate::{ArxivId, Ref};
57
58const SOURCE_KEY: &str = "ar5iv";
62
63pub const AR5IV_DEFAULT_BASE: &str = "https://ar5iv.labs.arxiv.org";
66
67const TEXT_CACHE_TTL_DAYS: i64 = 30;
71
72const TEXT_CACHE_SCHEMA_VERSION: &str = "1.0";
74
75#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
81#[serde(rename_all = "lowercase")]
82#[non_exhaustive]
83pub enum TextSource {
84 Ar5iv,
86}
87
88#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
90pub struct TextSection {
91 pub heading: Option<String>,
94 pub text: String,
97}
98
99#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
105pub struct PaperText {
106 pub arxiv_id: String,
108 pub source: TextSource,
110 pub title: Option<String>,
112 pub sections: Vec<TextSection>,
114 pub char_count: usize,
116 pub truncated: bool,
118 pub retrieved_from: String,
121}
122
123pub async fn paper_text(
152 base: &Url,
153 id: &ArxivId,
154 max_chars: Option<usize>,
155 ctx: &FetchContext,
156) -> Result<PaperText, FetchError> {
157 if let Some(root) = &ctx.cache_root {
160 if let Some(full) = cache_read(root, id) {
161 return Ok(apply_max_chars(full, max_chars));
162 }
163 }
164
165 let full = fetch_and_parse(base, id, ctx).await?;
166
167 if let Some(root) = &ctx.cache_root {
168 cache_write(root, id, &full);
169 }
170
171 Ok(apply_max_chars(full, max_chars))
172}
173
174async fn fetch_and_parse(
177 base: &Url,
178 id: &ArxivId,
179 ctx: &FetchContext,
180) -> Result<PaperText, FetchError> {
181 let _permit = ctx.rate_limiter.acquire(SOURCE_KEY).await;
183
184 let url = ar5iv_url(base, id)?;
185 let (body, final_url) = ctx.http.fetch_bytes(SOURCE_KEY, url).await?;
188
189 let (title, sections) = parse_ar5iv(&body)?;
190
191 let char_count: usize = sections.iter().map(|s| s.text.chars().count()).sum();
204 if char_count == 0 {
205 return Err(FetchError::TextUnavailable {
206 arxiv_id: id.clone(),
207 });
208 }
209
210 let canonical = Ref::Arxiv(id.clone())
212 .promote(SOURCE_KEY, None)
213 .digest_hex();
214 ctx.log.append(RowInput {
215 event: LogEvent::Fetch,
216 result: LogResult::Ok,
217 capability: Capability::Oa,
220 ref_: Some(id.as_str()),
221 source: Some(SOURCE_KEY),
222 error_code: None,
223 size_bytes: Some(body.len() as u64),
224 license: Some("arxiv-default"),
225 store_path: None,
226 canonical_digest: Some(&canonical),
227 })?;
228
229 Ok(PaperText {
230 arxiv_id: id.as_str().to_string(),
231 source: TextSource::Ar5iv,
232 title,
233 sections,
234 char_count,
237 truncated: false,
238 retrieved_from: final_url.to_string(),
239 })
240}
241
242fn ar5iv_url(base: &Url, id: &ArxivId) -> Result<Url, FetchError> {
249 base.join(&format!("/html/{}", id.as_str()))
250 .map_err(|e| FetchError::SourceSchema {
251 hint: format!("ar5iv URL construction failed: {e}"),
252 })
253}
254
255fn apply_max_chars(full: PaperText, max_chars: Option<usize>) -> PaperText {
259 let Some(max) = max_chars else {
260 return full;
261 };
262
263 let mut out: Vec<TextSection> = Vec::new();
264 let mut used = 0usize;
265 let mut truncated = false;
266 for sec in full.sections {
267 if used >= max {
268 truncated = true;
269 break;
270 }
271 let remaining = max - used;
272 let len = sec.text.chars().count();
273 if len <= remaining {
274 used += len;
275 out.push(sec);
276 } else {
277 let cut: String = sec.text.chars().take(remaining).collect();
278 used += remaining;
279 out.push(TextSection {
280 heading: sec.heading,
281 text: cut,
282 });
283 truncated = true;
284 break;
285 }
286 }
287
288 PaperText {
289 arxiv_id: full.arxiv_id,
290 source: full.source,
291 title: full.title,
292 sections: out,
293 char_count: used,
294 truncated,
295 retrieved_from: full.retrieved_from,
296 }
297}
298
299fn is_skip_element(local: &str) -> bool {
307 matches!(local, "script" | "style" | "math")
308}
309
310fn heading_level(local: &str) -> Option<u8> {
312 match local {
313 "h1" => Some(1),
314 "h2" => Some(2),
315 "h3" => Some(3),
316 "h4" => Some(4),
317 "h5" => Some(5),
318 "h6" => Some(6),
319 _ => None,
320 }
321}
322
323fn extract_alttext(e: &quick_xml::events::BytesStart<'_>) -> Option<String> {
325 for attr in e.attributes().flatten() {
326 if attr.key.as_ref() == "alttext" {
327 if let Ok(v) = attr.normalized_value(quick_xml::XmlVersion::Explicit1_0) {
328 let s = v.into_owned();
329 if !s.trim().is_empty() {
330 return Some(s);
331 }
332 }
333 }
334 }
335 None
336}
337
338fn normalize(s: &str) -> String {
342 s.split_whitespace().collect::<Vec<_>>().join(" ")
343}
344
345fn parse_ar5iv(html: &[u8]) -> Result<(Option<String>, Vec<TextSection>), FetchError> {
359 let mut reader = Reader::from_reader(html);
360 let config = reader.config_mut();
361 config.trim_text(true);
362 config.check_end_names = false;
367
368 let mut sections: Vec<TextSection> = Vec::new();
369 let mut cur_heading: Option<String> = None;
370 let mut cur_text = String::new();
371
372 let mut title: Option<String> = None;
373 let mut title_buf = String::new();
374 let mut in_title = false;
375
376 let mut skip: u32 = 0;
380 let mut in_heading: u8 = 0;
382 let mut heading_buf = String::new();
383
384 let mut buf: Vec<u8> = Vec::new();
385 loop {
386 match reader.read_event_into(&mut buf) {
387 Ok(Event::Start(e)) => {
388 let name = e.name();
389 let local = local_name(name.as_ref());
390 if is_skip_element(local) {
391 if skip == 0 && local == "math" {
392 if let Some(alt) = extract_alttext(&e) {
393 let frag = format!("\\({alt}\\) ");
394 push_target(
395 in_title,
396 in_heading,
397 &mut title_buf,
398 &mut heading_buf,
399 &mut cur_text,
400 &frag,
401 );
402 }
403 }
404 skip += 1;
405 } else if let Some(level) = heading_level(local) {
406 flush_section(&mut sections, &mut cur_heading, &mut cur_text);
408 in_heading = level;
409 heading_buf.clear();
410 } else if local == "title" && title.is_none() {
411 in_title = true;
412 title_buf.clear();
413 }
414 buf.clear();
415 }
416 Ok(Event::Empty(e)) => {
417 let name = e.name();
418 let local = local_name(name.as_ref());
419 if skip == 0 && local == "math" {
422 if let Some(alt) = extract_alttext(&e) {
423 let frag = format!("\\({alt}\\) ");
424 push_target(
425 in_title,
426 in_heading,
427 &mut title_buf,
428 &mut heading_buf,
429 &mut cur_text,
430 &frag,
431 );
432 }
433 }
434 buf.clear();
435 }
436 Ok(Event::Text(t)) => {
437 match quick_xml::escape::unescape(&t).ok().map(|c| c.into_owned()) {
438 Some(s) => {
439 if !s.is_empty() && skip == 0 {
440 let mut frag = s;
441 frag.push(' ');
442 push_target(
443 in_title,
444 in_heading,
445 &mut title_buf,
446 &mut heading_buf,
447 &mut cur_text,
448 &frag,
449 );
450 }
451 }
452 None => {
457 tracing::debug!(
458 "ar5iv: skipped a text fragment that failed to decode/unescape"
459 )
460 }
461 }
462 buf.clear();
463 }
464 Ok(Event::End(e)) => {
465 let name = e.name();
466 let local = local_name(name.as_ref());
467 if is_skip_element(local) {
468 skip = skip.saturating_sub(1);
469 } else if heading_level(local).is_some() && in_heading > 0 {
470 cur_heading = {
471 let h = normalize(&heading_buf);
472 if h.is_empty() {
473 None
474 } else {
475 Some(h)
476 }
477 };
478 in_heading = 0;
479 cur_text.clear();
481 } else if local == "title" && in_title {
482 in_title = false;
483 let t = normalize(&title_buf);
484 if !t.is_empty() {
485 title = Some(t);
486 }
487 }
488 buf.clear();
489 }
490 Ok(Event::Eof) => break,
491 Err(e) => {
492 tracing::debug!(error = %e, "ar5iv HTML parse error; returning best-effort partial text");
498 break;
499 }
500 _ => {
501 buf.clear();
502 }
503 }
504 }
505
506 flush_section(&mut sections, &mut cur_heading, &mut cur_text);
509
510 Ok((title, sections))
511}
512
513fn push_target(
516 in_title: bool,
517 in_heading: u8,
518 title_buf: &mut String,
519 heading_buf: &mut String,
520 cur_text: &mut String,
521 frag: &str,
522) {
523 if in_title {
524 title_buf.push_str(frag);
525 } else if in_heading > 0 {
526 heading_buf.push_str(frag);
527 } else {
528 cur_text.push_str(frag);
529 }
530}
531
532fn flush_section(
536 sections: &mut Vec<TextSection>,
537 cur_heading: &mut Option<String>,
538 cur_text: &mut String,
539) {
540 let text = normalize(cur_text);
541 if !text.is_empty() || cur_heading.is_some() {
542 sections.push(TextSection {
543 heading: cur_heading.clone(),
544 text,
545 });
546 }
547 cur_text.clear();
548}
549
550fn local_name(qname: &str) -> &str {
553 match qname.rfind(':') {
554 Some(idx) => &qname[idx + 1..],
555 None => qname,
556 }
557}
558
559#[derive(Debug, Serialize, Deserialize)]
566struct TextCacheEntry {
567 schema_version: String,
568 fetched_at: String,
570 ttl_seconds: i64,
571 paper_text: PaperText,
572}
573
574fn cache_file(cache_root: &Utf8Path, id: &ArxivId) -> Utf8PathBuf {
577 let safekey = Ref::Arxiv(id.clone()).safekey();
578 cache_root
579 .join("text")
580 .join(format!("{}.json", safekey.as_str()))
581}
582
583fn cache_read(cache_root: &Utf8Path, id: &ArxivId) -> Option<PaperText> {
586 cache_read_at(cache_root, id, Utc::now())
587}
588
589fn cache_read_at(cache_root: &Utf8Path, id: &ArxivId, now: DateTime<Utc>) -> Option<PaperText> {
591 let path = cache_file(cache_root, id);
592 let text = std::fs::read_to_string(&path).ok()?;
593 let entry: TextCacheEntry = serde_json::from_str(&text).ok()?;
594 let fetched: DateTime<Utc> = DateTime::parse_from_rfc3339(&entry.fetched_at)
595 .ok()?
596 .with_timezone(&Utc);
597 if now > fetched + Duration::seconds(entry.ttl_seconds) {
598 return None;
599 }
600 Some(entry.paper_text)
601}
602
603fn cache_write(cache_root: &Utf8Path, id: &ArxivId, full: &PaperText) -> bool {
607 cache_write_at(cache_root, id, full, Utc::now())
608}
609
610fn cache_write_at(
612 cache_root: &Utf8Path,
613 id: &ArxivId,
614 full: &PaperText,
615 now: DateTime<Utc>,
616) -> bool {
617 let entry = TextCacheEntry {
618 schema_version: TEXT_CACHE_SCHEMA_VERSION.to_string(),
619 fetched_at: now.to_rfc3339(),
620 ttl_seconds: TEXT_CACHE_TTL_DAYS * 86_400,
621 paper_text: full.clone(),
622 };
623 let json = match serde_json::to_string(&entry) {
624 Ok(s) => s,
625 Err(e) => {
626 tracing::debug!(error = %e, "text cache: serialize failed; skipping write");
627 return false;
628 }
629 };
630 let path = cache_file(cache_root, id);
631 if let Some(parent) = path.parent() {
632 if let Err(e) = std::fs::create_dir_all(parent) {
633 tracing::debug!(error = %e, dir = %parent, "text cache: mkdir failed; skipping write");
634 return false;
635 }
636 }
637 if let Err(e) = std::fs::write(&path, json) {
638 tracing::debug!(error = %e, path = %path, "text cache: write failed");
639 return false;
640 }
641 true
642}
643
644#[cfg(test)]
649#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
650mod tests {
651 use super::*;
652
653 use std::sync::Arc;
654
655 use camino::Utf8PathBuf;
656 use tempfile::TempDir;
657 use wiremock::matchers::{method, path as path_matcher};
658 use wiremock::{Mock, MockServer, ResponseTemplate};
659
660 use crate::http::HttpClient;
661 use crate::provenance::{LogRow, ProvenanceLog};
662 use crate::rate_limiter::RateLimiter;
663 use crate::RateLimits;
664
665 const SAMPLE_AR5IV: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
669<!DOCTYPE html>
670<html xmlns="http://www.w3.org/1999/xhtml">
671<head>
672 <title>Tropical Tensor Networks</title>
673 <style>.ltx_page { color: black; }</style>
674</head>
675<body>
676 <div class="ltx_page_content">
677 <p>We study tropical tensor networks for spin glasses.</p>
678 <section class="ltx_section">
679 <h2 class="ltx_title">1 Introduction</h2>
680 <p>The free energy is <math alttext="F = -kT \log Z"><mrow><mi>F</mi></mrow></math> in the limit.</p>
681 <script>trackingPixel();</script>
682 </section>
683 <section class="ltx_section">
684 <h2 class="ltx_title">2 Methods</h2>
685 <p>We use a contraction scheme.</p>
686 </section>
687 </div>
688</body>
689</html>"#;
690
691 fn build_test_context(
692 wiremock_host: &str,
693 cache: Option<Utf8PathBuf>,
694 ) -> (TempDir, FetchContext) {
695 let td = TempDir::new().expect("tempdir");
696 let log_dir =
697 Utf8PathBuf::try_from(td.path().to_path_buf()).expect("temp dir path must be UTF-8");
698 let log_path = log_dir.join("test.jsonl");
699
700 let http = Arc::new(HttpClient::new_for_tests_allow_http("ar5iv", wiremock_host));
701 let rate_limiter = Arc::new(RateLimiter::new(RateLimits::HARD_CODED));
702 let session_id = "01J0000000000000000000TEST".to_string();
703 let log = Arc::new(
704 ProvenanceLog::open(log_path, session_id.clone()).expect("provenance log opens"),
705 );
706 let ctx = FetchContext {
707 http,
708 rate_limiter,
709 log,
710 session_id,
711 cache_root: cache,
712 };
713 (td, ctx)
714 }
715
716 fn read_rows(path: &camino::Utf8Path) -> Vec<LogRow> {
717 let raw = std::fs::read_to_string(path).expect("read log");
718 raw.lines()
719 .filter(|l| !l.is_empty())
720 .map(|l| serde_json::from_str::<LogRow>(l).expect("valid LogRow"))
721 .collect()
722 }
723
724 #[test]
727 fn parse_extracts_title_sections_and_inline_math() {
728 let (title, sections) = parse_ar5iv(SAMPLE_AR5IV.as_bytes()).expect("parses");
729 assert_eq!(title.as_deref(), Some("Tropical Tensor Networks"));
730 assert_eq!(
731 sections.len(),
732 3,
733 "lead + two headed sections: {sections:?}"
734 );
735
736 assert_eq!(sections[0].heading, None);
738 assert_eq!(
739 sections[0].text,
740 "We study tropical tensor networks for spin glasses."
741 );
742
743 assert_eq!(sections[1].heading.as_deref(), Some("1 Introduction"));
744 assert_eq!(
747 sections[1].text,
748 "The free energy is \\(F = -kT \\log Z\\) in the limit."
749 );
750 assert!(
751 !sections[1].text.contains("trackingPixel"),
752 "script content must be skipped: {}",
753 sections[1].text
754 );
755
756 assert_eq!(sections[2].heading.as_deref(), Some("2 Methods"));
757 assert_eq!(sections[2].text, "We use a contraction scheme.");
758 }
759
760 #[test]
761 fn parse_no_headings_yields_single_lead_section() {
762 let xml = r#"<html><body><p>One paragraph only.</p><p>Second one.</p></body></html>"#;
763 let (title, sections) = parse_ar5iv(xml.as_bytes()).expect("parses");
764 assert!(title.is_none());
765 assert_eq!(sections.len(), 1);
766 assert_eq!(sections[0].heading, None);
767 assert_eq!(sections[0].text, "One paragraph only. Second one.");
768 }
769
770 #[test]
771 fn parse_empty_body_yields_nothing() {
772 let xml = r#"<html><head></head><body></body></html>"#;
773 let (title, sections) = parse_ar5iv(xml.as_bytes()).expect("parses");
774 assert!(title.is_none());
775 assert!(sections.is_empty());
776 }
777
778 #[test]
779 fn parse_mismatched_tags_recovers_full_document() {
780 let xml = r#"<html><body><p>Alpha beta <b>bold</p><h2>Sec</h2><p>Body text</body>"#;
784 let res = parse_ar5iv(xml.as_bytes());
785 assert!(res.is_ok(), "best-effort parse must not error: {res:?}");
786 let (_title, sections) = res.expect("ok");
787 let joined: String = sections
788 .iter()
789 .map(|s| s.text.as_str())
790 .collect::<Vec<_>>()
791 .join(" ");
792 assert!(
793 joined.contains("Body text") && joined.contains("bold"),
794 "recovered text past the mismatched tags: {joined:?}"
795 );
796 }
797
798 #[test]
799 fn parse_hard_syntax_error_degrades_to_partial_not_error() {
800 let xml = r#"<html><body><p>Prefix kept here</p><p>Bad & entity halts</body>"#;
805 let res = parse_ar5iv(xml.as_bytes());
806 assert!(res.is_ok(), "hard syntax error must NOT error: {res:?}");
807 let (_title, sections) = res.expect("ok");
808 let joined: String = sections
809 .iter()
810 .map(|s| s.text.as_str())
811 .collect::<Vec<_>>()
812 .join(" ");
813 assert!(
814 joined.contains("Prefix kept here"),
815 "the prefix before the hard error is retained: {joined:?}"
816 );
817 }
818
819 fn full_fixture() -> PaperText {
822 PaperText {
823 arxiv_id: "2401.12345".into(),
824 source: TextSource::Ar5iv,
825 title: Some("T".into()),
826 sections: vec![
827 TextSection {
828 heading: None,
829 text: "abcde".into(),
830 }, TextSection {
832 heading: Some("H".into()),
833 text: "fghij".into(),
834 }, ],
836 char_count: 10,
837 truncated: false,
838 retrieved_from: "https://ar5iv.labs.arxiv.org/html/2401.12345".into(),
839 }
840 }
841
842 #[test]
843 fn max_chars_none_returns_full() {
844 let out = apply_max_chars(full_fixture(), None);
845 assert!(!out.truncated);
846 assert_eq!(out.char_count, 10);
847 assert_eq!(out.sections.len(), 2);
848 }
849
850 #[test]
851 fn max_chars_above_total_is_untruncated() {
852 let out = apply_max_chars(full_fixture(), Some(100));
853 assert!(!out.truncated);
854 assert_eq!(out.char_count, 10);
855 }
856
857 #[test]
858 fn max_chars_cuts_within_a_section() {
859 let out = apply_max_chars(full_fixture(), Some(7));
861 assert!(out.truncated);
862 assert_eq!(out.char_count, 7);
863 assert_eq!(out.sections.len(), 2);
864 assert_eq!(out.sections[1].text, "fg");
865 assert_eq!(out.sections[1].heading.as_deref(), Some("H"));
866 }
867
868 #[test]
869 fn max_chars_drops_trailing_sections_on_exact_boundary() {
870 let out = apply_max_chars(full_fixture(), Some(5));
872 assert!(out.truncated);
873 assert_eq!(out.char_count, 5);
874 assert_eq!(out.sections.len(), 1);
875 }
876
877 #[test]
878 fn max_chars_zero_yields_no_text_but_flags_truncated() {
879 let out = apply_max_chars(full_fixture(), Some(0));
880 assert!(out.truncated);
881 assert_eq!(out.char_count, 0);
882 assert!(out.sections.is_empty());
883 }
884
885 #[test]
886 fn max_chars_truncation_is_char_boundary_safe_for_multibyte() {
887 let full = PaperText {
890 arxiv_id: "2401.12345".into(),
891 source: TextSource::Ar5iv,
892 title: None,
893 sections: vec![TextSection {
894 heading: None,
895 text: "あいうえお漢字".into(),
896 }],
897 char_count: 7,
898 truncated: false,
899 retrieved_from: "u".into(),
900 };
901 let out = apply_max_chars(full, Some(3));
902 assert!(out.truncated);
903 assert_eq!(out.char_count, 3);
904 assert_eq!(out.sections[0].text, "あいう");
905 }
906
907 #[test]
910 fn ar5iv_url_new_and_old_style() {
911 let base = Url::parse(AR5IV_DEFAULT_BASE).expect("base");
912 let new = ar5iv_url(&base, &ArxivId::parse("2401.12345").unwrap()).expect("url");
913 assert_eq!(new.path(), "/html/2401.12345");
914 assert_eq!(new.host_str(), Some("ar5iv.labs.arxiv.org"));
915 let old = ar5iv_url(&base, &ArxivId::parse("cond-mat/9501001").unwrap()).expect("url");
916 assert_eq!(old.path(), "/html/cond-mat/9501001");
917 }
918
919 #[test]
922 fn cache_write_then_read_round_trips() {
923 let dir = TempDir::new().unwrap();
924 let root = Utf8Path::from_path(dir.path()).unwrap();
925 let id = ArxivId::parse("2401.12345").unwrap();
926 let now = Utc::now();
927 assert!(cache_write_at(root, &id, &full_fixture(), now));
928 let got = cache_read_at(root, &id, now).expect("cache hit");
929 assert_eq!(got.arxiv_id, "2401.12345");
930 assert_eq!(got.sections.len(), 2);
931 assert!(!got.truncated, "cache stores the full, untruncated text");
932 }
933
934 #[test]
935 fn cache_miss_when_expired() {
936 let dir = TempDir::new().unwrap();
937 let root = Utf8Path::from_path(dir.path()).unwrap();
938 let id = ArxivId::parse("2401.12345").unwrap();
939 let written = Utc::now();
940 assert!(cache_write_at(root, &id, &full_fixture(), written));
941 let later = written + Duration::days(TEXT_CACHE_TTL_DAYS + 1);
942 assert!(cache_read_at(root, &id, later).is_none());
943 }
944
945 #[test]
946 fn cache_file_path_uses_text_dir_and_safekey() {
947 let root = Utf8Path::new("/tmp/cache");
948 let id = ArxivId::parse("2401.12345").unwrap();
949 let p = cache_file(root, &id);
950 assert!(p.components().any(|c| c.as_str() == "text"));
951 assert!(p.as_str().ends_with(".json"));
952 }
953
954 #[tokio::test]
957 async fn paper_text_fetches_parses_and_logs() {
958 let server = MockServer::start().await;
959 Mock::given(method("GET"))
960 .and(path_matcher("/html/2401.12345"))
961 .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
962 .mount(&server)
963 .await;
964
965 let host = server
966 .uri()
967 .parse::<Url>()
968 .unwrap()
969 .host_str()
970 .unwrap()
971 .to_string();
972 let (_td, ctx) = build_test_context(&host, None);
973 let log_path = ctx.log.path().to_path_buf();
974 let base = Url::parse(&server.uri()).expect("wiremock URI parses");
975 let id = ArxivId::parse("2401.12345").unwrap();
976
977 let out = paper_text(&base, &id, None, &ctx).await.expect("ok");
978 assert_eq!(out.arxiv_id, "2401.12345");
979 assert_eq!(out.source, TextSource::Ar5iv);
980 assert_eq!(out.title.as_deref(), Some("Tropical Tensor Networks"));
981 assert_eq!(out.sections.len(), 3);
982 assert!(!out.truncated);
983
984 let rows = read_rows(&log_path);
986 assert_eq!(rows.len(), 1, "one fetch row expected");
987 assert_eq!(rows[0].source.as_deref(), Some("ar5iv"));
988 assert_eq!(rows[0].ref_.as_deref(), Some("2401.12345"));
989 assert!(rows[0].error_code.is_none());
990 }
991
992 #[tokio::test]
993 async fn paper_text_truncates_when_max_chars_set() {
994 let server = MockServer::start().await;
995 Mock::given(method("GET"))
996 .and(path_matcher("/html/2401.12345"))
997 .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
998 .mount(&server)
999 .await;
1000 let host = server
1001 .uri()
1002 .parse::<Url>()
1003 .unwrap()
1004 .host_str()
1005 .unwrap()
1006 .to_string();
1007 let (_td, ctx) = build_test_context(&host, None);
1008 let base = Url::parse(&server.uri()).expect("uri");
1009 let id = ArxivId::parse("2401.12345").unwrap();
1010
1011 let out = paper_text(&base, &id, Some(10), &ctx).await.expect("ok");
1012 assert!(out.truncated);
1013 assert_eq!(out.char_count, 10);
1014 }
1015
1016 #[tokio::test]
1017 async fn paper_text_second_call_is_served_from_cache() {
1018 let server = MockServer::start().await;
1022 Mock::given(method("GET"))
1023 .and(path_matcher("/html/2401.12345"))
1024 .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
1025 .up_to_n_times(1)
1026 .mount(&server)
1027 .await;
1028
1029 let host = server
1030 .uri()
1031 .parse::<Url>()
1032 .unwrap()
1033 .host_str()
1034 .unwrap()
1035 .to_string();
1036 let cache_dir = TempDir::new().unwrap();
1037 let cache_root =
1038 Utf8PathBuf::try_from(cache_dir.path().to_path_buf()).expect("utf8 cache root");
1039 let (_td, ctx) = build_test_context(&host, Some(cache_root));
1040 let base = Url::parse(&server.uri()).expect("uri");
1041 let id = ArxivId::parse("2401.12345").unwrap();
1042
1043 let first = paper_text(&base, &id, None, &ctx).await.expect("first ok");
1044 assert_eq!(first.sections.len(), 3);
1045 let second = paper_text(&base, &id, None, &ctx)
1047 .await
1048 .expect("second call served from cache");
1049 assert_eq!(second.sections.len(), 3);
1050 assert_eq!(second.title, first.title);
1051 }
1052
1053 #[tokio::test]
1054 async fn paper_text_empty_body_is_text_unavailable() {
1055 let server = MockServer::start().await;
1056 Mock::given(method("GET"))
1057 .and(path_matcher("/html/2401.99999"))
1058 .respond_with(
1059 ResponseTemplate::new(200)
1060 .set_body_string("<html><head></head><body></body></html>"),
1061 )
1062 .mount(&server)
1063 .await;
1064 let host = server
1065 .uri()
1066 .parse::<Url>()
1067 .unwrap()
1068 .host_str()
1069 .unwrap()
1070 .to_string();
1071 let (_td, ctx) = build_test_context(&host, None);
1072 let base = Url::parse(&server.uri()).expect("uri");
1073 let id = ArxivId::parse("2401.99999").unwrap();
1074
1075 let err = paper_text(&base, &id, None, &ctx)
1080 .await
1081 .expect_err("empty body must be TextUnavailable");
1082 assert!(
1083 matches!(err, FetchError::TextUnavailable { .. }),
1084 "got {err:?}"
1085 );
1086 assert_eq!(
1087 crate::ErrorCode::from(&err),
1088 crate::ErrorCode::TextUnavailable
1089 );
1090 }
1091
1092 #[tokio::test]
1093 async fn paper_text_prose_free_body_is_text_unavailable() {
1094 let server = MockServer::start().await;
1101 Mock::given(method("GET"))
1102 .and(path_matcher("/html/2012.03644"))
1103 .respond_with(
1104 ResponseTemplate::new(200).set_body_string(
1105 "<html><head></head><body><h2>1 Introduction</h2></body></html>",
1106 ),
1107 )
1108 .mount(&server)
1109 .await;
1110 let host = server
1111 .uri()
1112 .parse::<Url>()
1113 .unwrap()
1114 .host_str()
1115 .unwrap()
1116 .to_string();
1117 let (_td, ctx) = build_test_context(&host, None);
1118 let base = Url::parse(&server.uri()).expect("uri");
1119 let id = ArxivId::parse("2012.03644").unwrap();
1120
1121 let err = paper_text(&base, &id, None, &ctx)
1122 .await
1123 .expect_err("a body with headings but no prose must be TextUnavailable");
1124 assert!(
1125 matches!(err, FetchError::TextUnavailable { .. }),
1126 "got {err:?}"
1127 );
1128 }
1129
1130 #[tokio::test]
1131 async fn paper_text_404_surfaces_http_error() {
1132 let server = MockServer::start().await;
1133 Mock::given(method("GET"))
1134 .and(path_matcher("/html/2401.00000"))
1135 .respond_with(ResponseTemplate::new(404))
1136 .mount(&server)
1137 .await;
1138 let host = server
1139 .uri()
1140 .parse::<Url>()
1141 .unwrap()
1142 .host_str()
1143 .unwrap()
1144 .to_string();
1145 let (_td, ctx) = build_test_context(&host, None);
1146 let base = Url::parse(&server.uri()).expect("uri");
1147 let id = ArxivId::parse("2401.00000").unwrap();
1148
1149 let err = paper_text(&base, &id, None, &ctx)
1150 .await
1151 .expect_err("404 must surface");
1152 assert_eq!(crate::ErrorCode::from(&err), crate::ErrorCode::NotFound);
1154 }
1155
1156 #[test]
1157 fn text_source_serializes_lowercase() {
1158 let s = serde_json::to_string(&TextSource::Ar5iv).expect("serialize");
1159 assert_eq!(s, "\"ar5iv\"");
1160 }
1161}