Skip to main content

doiget_core/
paper_text.rs

1//! Full-text **extraction** of an arXiv paper from ar5iv's LaTeXML XHTML
2//! (the #281 "read" step; ADR-0032).
3//!
4//! This is the step that lets an agent actually *read* a paper without an
5//! external pdf-to-text tool. It is deliberately **distinct from PDF
6//! content processing** (permanent non-goal #1 / ADR-0003): the PDF blob
7//! is never opened. Instead doiget fetches a *separate, already-structured
8//! artifact* — the publisher-rendered HTML — and extracts text from it.
9//! ADR-0032 D1 records the boundary: PDF-blob parsing and OCR stay
10//! permanently out of scope; structured HTML/XML full text is in scope.
11//!
12//! ## Source (ADR-0032 D3)
13//!
14//! PR4 ships one source, **ar5iv** (`ar5iv.labs.arxiv.org/html/<id>`),
15//! which renders arXiv papers as LaTeXML XHTML. The fetch goes through the
16//! dedicated `"ar5iv"` source key (see [`crate::http::fulltext_allowlist`])
17//! so the provenance trail distinguishes ar5iv full text from the arXiv
18//! PDF/Atom API. PMC / Europe PMC JATS is a planned follow-up source.
19//!
20//! ## Capability tier (ADR-0032 D2)
21//!
22//! Tier 1 OA metadata, **always-on**: no env gate, no Cargo feature gate.
23//! Read-only, open-access, never a PDF reinterpretation — same posture
24//! class as discovery search (ADR-0031). Ships in the default `oa-only`
25//! binary.
26//!
27//! ## Caching (ADR-0032 D4)
28//!
29//! Extracted text is cached at `<cache_root>/text/<safekey>.json` (the
30//! doiget-private cache root, `docs/CACHE.md`) — **not** the shared
31//! `~/papers/` store (`docs/STORE.md`), so no cross-tool coordination is
32//! needed. The cache holds the **full** text; `max_chars` truncation is a
33//! view applied on return, so one cached entry serves any `max_chars`.
34//! Best-effort: a miss / parse error / write failure degrades to a
35//! re-fetch, never an error (mirrors [`crate::resolver_cache`]).
36//!
37//! ## Extraction
38//!
39//! A `quick-xml` walk (the same parser the arXiv Atom path uses) splits
40//! the document into `{ heading, text }` sections on `h1`–`h6`, skips
41//! `script` / `style` / `math` subtrees — capturing each `<math>`'s
42//! `alttext` (the LaTeX source) as inline `\(…\)` text so formulae read
43//! cleanly rather than as MathML noise — and normalizes whitespace.
44//! Extraction is best-effort: it supplies the text it can and flags
45//! truncation; it does not promise faithful reconstruction.
46
47use camino::{Utf8Path, Utf8PathBuf};
48use chrono::{DateTime, Duration, Utc};
49use quick_xml::events::Event;
50use quick_xml::Reader;
51use serde::{Deserialize, Serialize};
52use url::Url;
53
54use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
55use crate::source::{FetchContext, FetchError};
56use crate::{ArxivId, Ref};
57
58/// Source key for the per-source HTTP client + redirect allowlist. Kept
59/// distinct from `"arxiv"` (PDF/Atom) so the provenance trail records that
60/// extracted text came from the ar5iv renderer (ADR-0032 D3).
61const SOURCE_KEY: &str = "ar5iv";
62
63/// Production ar5iv base. Overridable via `DOIGET_AR5IV_BASE` (test
64/// wiremock origin), mirroring the `DOIGET_ARXIV_BASE` override.
65pub const AR5IV_DEFAULT_BASE: &str = "https://ar5iv.labs.arxiv.org";
66
67/// Cache entry TTL. ar5iv output for a given id is effectively static, so
68/// a long TTL is safe; 30 days bounds staleness while keeping repeat reads
69/// offline.
70const TEXT_CACHE_TTL_DAYS: i64 = 30;
71
72/// On-disk text-cache schema version (`docs/CACHE.md`).
73const TEXT_CACHE_SCHEMA_VERSION: &str = "1.0";
74
75/// Which structured full-text source produced a [`PaperText`].
76///
77/// `#[non_exhaustive]` reserves room for future Tier-1 full-text sources
78/// (PMC / Europe PMC JATS) without a breaking change; the wire form is the
79/// lowercase variant name.
80#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
81#[serde(rename_all = "lowercase")]
82#[non_exhaustive]
83pub enum TextSource {
84    /// ar5iv LaTeXML XHTML (`ar5iv.labs.arxiv.org`). Serializes `"ar5iv"`.
85    Ar5iv,
86}
87
88/// One `{ heading, text }` section of an extracted paper.
89#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
90pub struct TextSection {
91    /// Section heading (`h1`–`h6` text), or `None` for the lead matter
92    /// before the first heading.
93    pub heading: Option<String>,
94    /// Plain section text: tags stripped, whitespace normalized, inline
95    /// math rendered as the LaTeX `\(…\)` from the source's `alttext`.
96    pub text: String,
97}
98
99/// Extracted full text of an arXiv paper.
100///
101/// The value returned by [`paper_text`] reflects the caller's `max_chars`
102/// (see [`PaperText::truncated`]); the cached form is always the full
103/// text.
104#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
105pub struct PaperText {
106    /// The arXiv id this text belongs to.
107    pub arxiv_id: String,
108    /// Which source produced the text (PR4: always [`TextSource::Ar5iv`]).
109    pub source: TextSource,
110    /// Document title, if one was extracted (the `<title>` element).
111    pub title: Option<String>,
112    /// Ordered sections.
113    pub sections: Vec<TextSection>,
114    /// Total `char`s across `sections[].text` (after any truncation).
115    pub char_count: usize,
116    /// `true` when the returned text was truncated to honor `max_chars`.
117    pub truncated: bool,
118    /// Final URL the text was retrieved from (after redirects), for
119    /// provenance. Carried through the cache.
120    pub retrieved_from: String,
121}
122
123/// Fetch and extract the full text of an arXiv paper from ar5iv.
124///
125/// `base` is the ar5iv base URL (production [`AR5IV_DEFAULT_BASE`]; tests
126/// inject a wiremock origin via `DOIGET_AR5IV_BASE`). `max_chars` caps the
127/// returned **section body text** (`char_count`; the `title` and section
128/// `heading`s are not counted against it) (`None` = no cap); truncation is
129/// flagged on
130/// [`PaperText::truncated`], never silent. When `ctx.cache_root` is `Some`,
131/// a fresh cache entry is served from disk; otherwise the text is fetched,
132/// parsed, cached (best-effort), and one `Fetch` provenance row is emitted.
133///
134/// Never opens a PDF — this is a separate fetch of the ar5iv HTML artifact
135/// (ADR-0032 D1).
136///
137/// # Errors
138///
139/// - [`FetchError::Http`] for transport / status failures (a 404 / 410
140///   collapses to `NOT_FOUND` at the boundary — the id is genuinely absent
141///   from ar5iv).
142/// - [`FetchError::TextUnavailable`] when ar5iv returns a **200** with no
143///   extractable prose (`char_count == 0`: the paper was never converted to
144///   HTML). Distinct from `NotFound`: the id is valid and the PDF may be
145///   fetchable, so an agent should fetch rather than treat the ref as wrong
146///   (issue #302).
147/// - [`FetchError::SourceSchema`] if the ar5iv URL cannot be constructed
148///   from `base` + the id. (HTML parsing itself is best-effort and
149///   infallible on content — see the `parse_ar5iv` helper.)
150/// - [`FetchError::Log`] if the provenance write fails (fail-closed).
151pub async fn paper_text(
152    base: &Url,
153    id: &ArxivId,
154    max_chars: Option<usize>,
155    ctx: &FetchContext,
156) -> Result<PaperText, FetchError> {
157    // Cache read (best-effort). A hit serves the full text from disk; the
158    // `max_chars` view is applied below, so the same entry serves any cap.
159    if let Some(root) = &ctx.cache_root {
160        if let Some(full) = cache_read(root, id) {
161            return Ok(apply_max_chars(full, max_chars));
162        }
163    }
164
165    let full = fetch_and_parse(base, id, ctx).await?;
166
167    if let Some(root) = &ctx.cache_root {
168        cache_write(root, id, &full);
169    }
170
171    Ok(apply_max_chars(full, max_chars))
172}
173
174/// Network fetch + parse + provenance for one ar5iv document. Always
175/// returns the **full** (untruncated) text.
176async fn fetch_and_parse(
177    base: &Url,
178    id: &ArxivId,
179    ctx: &FetchContext,
180) -> Result<PaperText, FetchError> {
181    // Politeness gate — same channel every source uses.
182    let _permit = ctx.rate_limiter.acquire(SOURCE_KEY).await;
183
184    let url = ar5iv_url(base, id)?;
185    // `fetch_bytes` (not `fetch_pdf`): the body is HTML, and the PDF
186    // magic-byte check must NOT apply here.
187    let (body, final_url) = ctx.http.fetch_bytes(SOURCE_KEY, url).await?;
188
189    let (title, sections) = parse_ar5iv(&body)?;
190
191    // A 200 with no readable prose (ar5iv's "not converted" placeholder, or
192    // a stub that parses to section shells with empty bodies) is an
193    // authoritative "nothing to read". Gate on the total extractable char
194    // count — NOT just `sections.is_empty()` — so the empty-bodied-section
195    // case (issue #302's repro: a non-empty `sections` whose text is all
196    // blank) is caught too and never escapes as an empty `Ok` (a silent
197    // exit-0 the caller misreads as a bad identifier).
198    //
199    // Surfaced as `TextUnavailable`, NOT `NotFound`: the arXiv id was
200    // already validated and resolved, so the honest signal is "this
201    // representation is missing — fetch the PDF", not "the id does not
202    // exist" (which would send an agent down a wrong debugging path).
203    let char_count: usize = sections.iter().map(|s| s.text.chars().count()).sum();
204    if char_count == 0 {
205        return Err(FetchError::TextUnavailable {
206            arxiv_id: id.clone(),
207        });
208    }
209
210    // ADR-0021 §1 canonical digest under the "ar5iv" resolver profile.
211    let canonical = Ref::Arxiv(id.clone())
212        .promote(SOURCE_KEY, None)
213        .digest_hex();
214    ctx.log.append(RowInput {
215        event: LogEvent::Fetch,
216        result: LogResult::Ok,
217        // OA full-text content; same capability class the arXiv PDF leg
218        // uses. Never a PDF reinterpretation (ADR-0032 D1).
219        capability: Capability::Oa,
220        ref_: Some(id.as_str()),
221        source: Some(SOURCE_KEY),
222        error_code: None,
223        size_bytes: Some(body.len() as u64),
224        license: Some("arxiv-default"),
225        store_path: None,
226        canonical_digest: Some(&canonical),
227    })?;
228
229    Ok(PaperText {
230        arxiv_id: id.as_str().to_string(),
231        source: TextSource::Ar5iv,
232        title,
233        sections,
234        // Computed above (the non-empty gate); reused here so the count is
235        // derived exactly once.
236        char_count,
237        truncated: false,
238        retrieved_from: final_url.to_string(),
239    })
240}
241
242/// Build the ar5iv URL for an id: `<base>/html/<id>`.
243///
244/// Old-style ids (`cond-mat/9501001`) contain a `/`; the resulting path
245/// `/html/cond-mat/9501001` is the form ar5iv expects. `Url::join` of the
246/// absolute reference `/html/<id>` resolves correctly for both id shapes
247/// (the base has no path beyond `/`), mirroring the arXiv PDF URL builder.
248fn ar5iv_url(base: &Url, id: &ArxivId) -> Result<Url, FetchError> {
249    base.join(&format!("/html/{}", id.as_str()))
250        .map_err(|e| FetchError::SourceSchema {
251            hint: format!("ar5iv URL construction failed: {e}"),
252        })
253}
254
255/// Apply a `max_chars` cap to a full [`PaperText`], producing the returned
256/// view. Truncation is by `char` (boundary-safe) and is flagged; sections
257/// past the budget are dropped and the straddling section is cut.
258fn apply_max_chars(full: PaperText, max_chars: Option<usize>) -> PaperText {
259    let Some(max) = max_chars else {
260        return full;
261    };
262
263    let mut out: Vec<TextSection> = Vec::new();
264    let mut used = 0usize;
265    let mut truncated = false;
266    for sec in full.sections {
267        if used >= max {
268            truncated = true;
269            break;
270        }
271        let remaining = max - used;
272        let len = sec.text.chars().count();
273        if len <= remaining {
274            used += len;
275            out.push(sec);
276        } else {
277            let cut: String = sec.text.chars().take(remaining).collect();
278            used += remaining;
279            out.push(TextSection {
280                heading: sec.heading,
281                text: cut,
282            });
283            truncated = true;
284            break;
285        }
286    }
287
288    PaperText {
289        arxiv_id: full.arxiv_id,
290        source: full.source,
291        title: full.title,
292        sections: out,
293        char_count: used,
294        truncated,
295        retrieved_from: full.retrieved_from,
296    }
297}
298
299// ---------------------------------------------------------------------------
300// ar5iv XHTML parser
301// ---------------------------------------------------------------------------
302
303/// Local element names whose entire subtree is skipped (text discarded).
304/// `math` is in this set, but its `alttext` is captured before the subtree
305/// is skipped (see [`extract_alttext`]).
306fn is_skip_element(local: &str) -> bool {
307    matches!(local, "script" | "style" | "math")
308}
309
310/// Heading level (1..=6) for an `h1`–`h6` local name, else `None`.
311fn heading_level(local: &str) -> Option<u8> {
312    match local {
313        "h1" => Some(1),
314        "h2" => Some(2),
315        "h3" => Some(3),
316        "h4" => Some(4),
317        "h5" => Some(5),
318        "h6" => Some(6),
319        _ => None,
320    }
321}
322
323/// Extract a `<math alttext="...">` LaTeX source, if present.
324fn extract_alttext(e: &quick_xml::events::BytesStart<'_>) -> Option<String> {
325    for attr in e.attributes().flatten() {
326        if attr.key.as_ref() == "alttext" {
327            if let Ok(v) = attr.normalized_value(quick_xml::XmlVersion::Explicit1_0) {
328                let s = v.into_owned();
329                if !s.trim().is_empty() {
330                    return Some(s);
331                }
332            }
333        }
334    }
335    None
336}
337
338/// Normalize collected text: collapse all whitespace runs to single spaces
339/// and trim. Keeps inline LaTeX (`\(…\)`) intact bar internal whitespace
340/// collapse.
341fn normalize(s: &str) -> String {
342    s.split_whitespace().collect::<Vec<_>>().join(" ")
343}
344
345/// Parse an ar5iv LaTeXML-XHTML body into `(title, sections)`.
346///
347/// Best-effort: see the module docs. Splits on `h1`–`h6`; skips
348/// `script` / `style` / `math` subtrees (capturing `math` `alttext` as
349/// inline `\(…\)`); normalizes whitespace. Sections with empty text but a
350/// heading are kept (a heading with no body still maps the document).
351///
352/// Best-effort and **infallible** on content: a malformed/non-well-formed
353/// document is recovered (the reader runs with `check_end_names = false`,
354/// and a hard syntax error stops the walk while keeping the partial
355/// result rather than discarding it — ADR-0032 D3). An empty result is
356/// handled by the caller (→ `TextUnavailable`). Returns `Result` only to keep the
357/// call-site uniform; it does not currently produce an `Err`.
358fn parse_ar5iv(html: &[u8]) -> Result<(Option<String>, Vec<TextSection>), FetchError> {
359    let mut reader = Reader::from_reader(html);
360    let config = reader.config_mut();
361    config.trim_text(true);
362    // Best-effort (ADR-0032 D3): ar5iv LaTeXML output is normally
363    // well-formed XHTML, but real renderings can carry mismatched/void
364    // tags. Don't reject the whole document over an end-name mismatch —
365    // recover and keep extracting.
366    config.check_end_names = false;
367
368    let mut sections: Vec<TextSection> = Vec::new();
369    let mut cur_heading: Option<String> = None;
370    let mut cur_text = String::new();
371
372    let mut title: Option<String> = None;
373    let mut title_buf = String::new();
374    let mut in_title = false;
375
376    // Depth-counted skip: >0 means we are inside a script/style/math
377    // subtree and must discard text. Increment on the skip element's
378    // Start, decrement on its End. (`math` captures alttext first.)
379    let mut skip: u32 = 0;
380    // >0 means we are inside an h1..h6 element; text routes to heading_buf.
381    let mut in_heading: u8 = 0;
382    let mut heading_buf = String::new();
383
384    let mut buf: Vec<u8> = Vec::new();
385    loop {
386        match reader.read_event_into(&mut buf) {
387            Ok(Event::Start(e)) => {
388                let name = e.name();
389                let local = local_name(name.as_ref());
390                if is_skip_element(local) {
391                    if skip == 0 && local == "math" {
392                        if let Some(alt) = extract_alttext(&e) {
393                            let frag = format!("\\({alt}\\) ");
394                            push_target(
395                                in_title,
396                                in_heading,
397                                &mut title_buf,
398                                &mut heading_buf,
399                                &mut cur_text,
400                                &frag,
401                            );
402                        }
403                    }
404                    skip += 1;
405                } else if let Some(level) = heading_level(local) {
406                    // A new heading closes the current section.
407                    flush_section(&mut sections, &mut cur_heading, &mut cur_text);
408                    in_heading = level;
409                    heading_buf.clear();
410                } else if local == "title" && title.is_none() {
411                    in_title = true;
412                    title_buf.clear();
413                }
414                buf.clear();
415            }
416            Ok(Event::Empty(e)) => {
417                let name = e.name();
418                let local = local_name(name.as_ref());
419                // A self-closing `<math .../>` contributes its alttext but
420                // has no subtree to skip.
421                if skip == 0 && local == "math" {
422                    if let Some(alt) = extract_alttext(&e) {
423                        let frag = format!("\\({alt}\\) ");
424                        push_target(
425                            in_title,
426                            in_heading,
427                            &mut title_buf,
428                            &mut heading_buf,
429                            &mut cur_text,
430                            &frag,
431                        );
432                    }
433                }
434                buf.clear();
435            }
436            Ok(Event::Text(t)) => {
437                match quick_xml::escape::unescape(&t).ok().map(|c| c.into_owned()) {
438                    Some(s) => {
439                        if !s.is_empty() && skip == 0 {
440                            let mut frag = s;
441                            frag.push(' ');
442                            push_target(
443                                in_title,
444                                in_heading,
445                                &mut title_buf,
446                                &mut heading_buf,
447                                &mut cur_text,
448                                &frag,
449                            );
450                        }
451                    }
452                    // A text fragment that fails to decode/unescape is
453                    // dropped (best-effort), but log it: a *systematic*
454                    // decode failure would otherwise vanish silently while
455                    // the extraction still "succeeds" (review #285).
456                    None => {
457                        tracing::debug!(
458                            "ar5iv: skipped a text fragment that failed to decode/unescape"
459                        )
460                    }
461                }
462                buf.clear();
463            }
464            Ok(Event::End(e)) => {
465                let name = e.name();
466                let local = local_name(name.as_ref());
467                if is_skip_element(local) {
468                    skip = skip.saturating_sub(1);
469                } else if heading_level(local).is_some() && in_heading > 0 {
470                    cur_heading = {
471                        let h = normalize(&heading_buf);
472                        if h.is_empty() {
473                            None
474                        } else {
475                            Some(h)
476                        }
477                    };
478                    in_heading = 0;
479                    // The body of the new section starts fresh.
480                    cur_text.clear();
481                } else if local == "title" && in_title {
482                    in_title = false;
483                    let t = normalize(&title_buf);
484                    if !t.is_empty() {
485                        title = Some(t);
486                    }
487                }
488                buf.clear();
489            }
490            Ok(Event::Eof) => break,
491            Err(e) => {
492                // Best-effort (ADR-0032 D3): a syntax error deep in the
493                // document must not discard everything already collected.
494                // Stop here and return the partial result — an empty result
495                // still maps to TextUnavailable upstream. The error is observable
496                // on stderr, never on the stdout JSON-RPC channel.
497                tracing::debug!(error = %e, "ar5iv HTML parse error; returning best-effort partial text");
498                break;
499            }
500            _ => {
501                buf.clear();
502            }
503        }
504    }
505
506    // Final section (text after the last heading, or the lead matter when
507    // the document had no headings at all).
508    flush_section(&mut sections, &mut cur_heading, &mut cur_text);
509
510    Ok((title, sections))
511}
512
513/// Append `frag` to whichever buffer is currently active: the `<title>`
514/// buffer, the heading buffer, or the section-body buffer.
515fn push_target(
516    in_title: bool,
517    in_heading: u8,
518    title_buf: &mut String,
519    heading_buf: &mut String,
520    cur_text: &mut String,
521    frag: &str,
522) {
523    if in_title {
524        title_buf.push_str(frag);
525    } else if in_heading > 0 {
526        heading_buf.push_str(frag);
527    } else {
528        cur_text.push_str(frag);
529    }
530}
531
532/// Push the current `(heading, text)` as a section if it carries anything
533/// (non-empty body OR a heading), then reset the body buffer. The heading
534/// is left in place — it is replaced when the next heading is parsed.
535fn flush_section(
536    sections: &mut Vec<TextSection>,
537    cur_heading: &mut Option<String>,
538    cur_text: &mut String,
539) {
540    let text = normalize(cur_text);
541    if !text.is_empty() || cur_heading.is_some() {
542        sections.push(TextSection {
543            heading: cur_heading.clone(),
544            text,
545        });
546    }
547    cur_text.clear();
548}
549
550/// Strip an XML namespace prefix, returning the local part
551/// (`"xhtml:p"` -> `"p"`). Mirrors the arXiv Atom parser's helper.
552fn local_name(qname: &str) -> &str {
553    match qname.rfind(':') {
554        Some(idx) => &qname[idx + 1..],
555        None => qname,
556    }
557}
558
559// ---------------------------------------------------------------------------
560// Text cache (docs/CACHE.md; ADR-0032 D4)
561// ---------------------------------------------------------------------------
562
563/// On-disk cache entry. `paper_text` is stored as nested JSON (not a
564/// string) since the whole entry is JSON.
565#[derive(Debug, Serialize, Deserialize)]
566struct TextCacheEntry {
567    schema_version: String,
568    /// RFC 3339 UTC timestamp of the fetch that produced this entry.
569    fetched_at: String,
570    ttl_seconds: i64,
571    paper_text: PaperText,
572}
573
574/// The on-disk path for an id's text cache entry:
575/// `<cache_root>/text/<safekey>.json`.
576fn cache_file(cache_root: &Utf8Path, id: &ArxivId) -> Utf8PathBuf {
577    let safekey = Ref::Arxiv(id.clone()).safekey();
578    cache_root
579        .join("text")
580        .join(format!("{}.json", safekey.as_str()))
581}
582
583/// Read the cached full text for `id` if present and within its TTL. Any
584/// miss condition (absent / unparsable / expired) returns `None`.
585fn cache_read(cache_root: &Utf8Path, id: &ArxivId) -> Option<PaperText> {
586    cache_read_at(cache_root, id, Utc::now())
587}
588
589/// [`cache_read`] with an injected clock for tests.
590fn cache_read_at(cache_root: &Utf8Path, id: &ArxivId, now: DateTime<Utc>) -> Option<PaperText> {
591    let path = cache_file(cache_root, id);
592    let text = std::fs::read_to_string(&path).ok()?;
593    let entry: TextCacheEntry = serde_json::from_str(&text).ok()?;
594    let fetched: DateTime<Utc> = DateTime::parse_from_rfc3339(&entry.fetched_at)
595        .ok()?
596        .with_timezone(&Utc);
597    if now > fetched + Duration::seconds(entry.ttl_seconds) {
598        return None;
599    }
600    Some(entry.paper_text)
601}
602
603/// Write the full text for `id` to the cache. Best-effort: returns `false`
604/// (after a `tracing::debug!`) on any failure rather than propagating — a
605/// cache write must never fail an extraction.
606fn cache_write(cache_root: &Utf8Path, id: &ArxivId, full: &PaperText) -> bool {
607    cache_write_at(cache_root, id, full, Utc::now())
608}
609
610/// [`cache_write`] with an injected clock for tests.
611fn cache_write_at(
612    cache_root: &Utf8Path,
613    id: &ArxivId,
614    full: &PaperText,
615    now: DateTime<Utc>,
616) -> bool {
617    let entry = TextCacheEntry {
618        schema_version: TEXT_CACHE_SCHEMA_VERSION.to_string(),
619        fetched_at: now.to_rfc3339(),
620        ttl_seconds: TEXT_CACHE_TTL_DAYS * 86_400,
621        paper_text: full.clone(),
622    };
623    let json = match serde_json::to_string(&entry) {
624        Ok(s) => s,
625        Err(e) => {
626            tracing::debug!(error = %e, "text cache: serialize failed; skipping write");
627            return false;
628        }
629    };
630    let path = cache_file(cache_root, id);
631    if let Some(parent) = path.parent() {
632        if let Err(e) = std::fs::create_dir_all(parent) {
633            tracing::debug!(error = %e, dir = %parent, "text cache: mkdir failed; skipping write");
634            return false;
635        }
636    }
637    if let Err(e) = std::fs::write(&path, json) {
638        tracing::debug!(error = %e, path = %path, "text cache: write failed");
639        return false;
640    }
641    true
642}
643
644// ---------------------------------------------------------------------------
645// Tests
646// ---------------------------------------------------------------------------
647
648#[cfg(test)]
649#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
650mod tests {
651    use super::*;
652
653    use std::sync::Arc;
654
655    use camino::Utf8PathBuf;
656    use tempfile::TempDir;
657    use wiremock::matchers::{method, path as path_matcher};
658    use wiremock::{Mock, MockServer, ResponseTemplate};
659
660    use crate::http::HttpClient;
661    use crate::provenance::{LogRow, ProvenanceLog};
662    use crate::rate_limiter::RateLimiter;
663    use crate::RateLimits;
664
665    /// Synthetic ar5iv-shaped XHTML. Hand-crafted (not a snapshot) to avoid
666    /// third-party redistribution; exercises title, lead matter, two
667    /// headed sections, inline math `alttext`, and skipped script/style.
668    const SAMPLE_AR5IV: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
669<!DOCTYPE html>
670<html xmlns="http://www.w3.org/1999/xhtml">
671<head>
672  <title>Tropical Tensor Networks</title>
673  <style>.ltx_page { color: black; }</style>
674</head>
675<body>
676  <div class="ltx_page_content">
677    <p>We study tropical tensor networks for spin glasses.</p>
678    <section class="ltx_section">
679      <h2 class="ltx_title">1 Introduction</h2>
680      <p>The free energy is <math alttext="F = -kT \log Z"><mrow><mi>F</mi></mrow></math> in the limit.</p>
681      <script>trackingPixel();</script>
682    </section>
683    <section class="ltx_section">
684      <h2 class="ltx_title">2 Methods</h2>
685      <p>We use a contraction scheme.</p>
686    </section>
687  </div>
688</body>
689</html>"#;
690
691    fn build_test_context(
692        wiremock_host: &str,
693        cache: Option<Utf8PathBuf>,
694    ) -> (TempDir, FetchContext) {
695        let td = TempDir::new().expect("tempdir");
696        let log_dir =
697            Utf8PathBuf::try_from(td.path().to_path_buf()).expect("temp dir path must be UTF-8");
698        let log_path = log_dir.join("test.jsonl");
699
700        let http = Arc::new(HttpClient::new_for_tests_allow_http("ar5iv", wiremock_host));
701        let rate_limiter = Arc::new(RateLimiter::new(RateLimits::HARD_CODED));
702        let session_id = "01J0000000000000000000TEST".to_string();
703        let log = Arc::new(
704            ProvenanceLog::open(log_path, session_id.clone()).expect("provenance log opens"),
705        );
706        let ctx = FetchContext {
707            http,
708            rate_limiter,
709            log,
710            session_id,
711            cache_root: cache,
712        };
713        (td, ctx)
714    }
715
716    fn read_rows(path: &camino::Utf8Path) -> Vec<LogRow> {
717        let raw = std::fs::read_to_string(path).expect("read log");
718        raw.lines()
719            .filter(|l| !l.is_empty())
720            .map(|l| serde_json::from_str::<LogRow>(l).expect("valid LogRow"))
721            .collect()
722    }
723
724    // ---- parse_ar5iv -----------------------------------------------------
725
726    #[test]
727    fn parse_extracts_title_sections_and_inline_math() {
728        let (title, sections) = parse_ar5iv(SAMPLE_AR5IV.as_bytes()).expect("parses");
729        assert_eq!(title.as_deref(), Some("Tropical Tensor Networks"));
730        assert_eq!(
731            sections.len(),
732            3,
733            "lead + two headed sections: {sections:?}"
734        );
735
736        // Lead matter (before the first heading) has no heading.
737        assert_eq!(sections[0].heading, None);
738        assert_eq!(
739            sections[0].text,
740            "We study tropical tensor networks for spin glasses."
741        );
742
743        assert_eq!(sections[1].heading.as_deref(), Some("1 Introduction"));
744        // Inline math becomes the LaTeX `alttext` as `\(…\)`; the MathML
745        // (`<mi>F</mi>`) subtree is skipped.
746        assert_eq!(
747            sections[1].text,
748            "The free energy is \\(F = -kT \\log Z\\) in the limit."
749        );
750        assert!(
751            !sections[1].text.contains("trackingPixel"),
752            "script content must be skipped: {}",
753            sections[1].text
754        );
755
756        assert_eq!(sections[2].heading.as_deref(), Some("2 Methods"));
757        assert_eq!(sections[2].text, "We use a contraction scheme.");
758    }
759
760    #[test]
761    fn parse_no_headings_yields_single_lead_section() {
762        let xml = r#"<html><body><p>One paragraph only.</p><p>Second one.</p></body></html>"#;
763        let (title, sections) = parse_ar5iv(xml.as_bytes()).expect("parses");
764        assert!(title.is_none());
765        assert_eq!(sections.len(), 1);
766        assert_eq!(sections[0].heading, None);
767        assert_eq!(sections[0].text, "One paragraph only. Second one.");
768    }
769
770    #[test]
771    fn parse_empty_body_yields_nothing() {
772        let xml = r#"<html><head></head><body></body></html>"#;
773        let (title, sections) = parse_ar5iv(xml.as_bytes()).expect("parses");
774        assert!(title.is_none());
775        assert!(sections.is_empty());
776    }
777
778    #[test]
779    fn parse_mismatched_tags_recovers_full_document() {
780        // Mismatched/unclosed tags (a `<b>` closed by `</p>`, an unclosed
781        // `<body>`) must NOT discard the document: `check_end_names = false`
782        // recovers and keeps extracting past them (ADR-0032 D3).
783        let xml = r#"<html><body><p>Alpha beta <b>bold</p><h2>Sec</h2><p>Body text</body>"#;
784        let res = parse_ar5iv(xml.as_bytes());
785        assert!(res.is_ok(), "best-effort parse must not error: {res:?}");
786        let (_title, sections) = res.expect("ok");
787        let joined: String = sections
788            .iter()
789            .map(|s| s.text.as_str())
790            .collect::<Vec<_>>()
791            .join(" ");
792        assert!(
793            joined.contains("Body text") && joined.contains("bold"),
794            "recovered text past the mismatched tags: {joined:?}"
795        );
796    }
797
798    #[test]
799    fn parse_hard_syntax_error_degrades_to_partial_not_error() {
800        // A bare `&` (invalid entity) is a hard reader error; best-effort
801        // (ADR-0032 D3) returns the partial text collected before it rather
802        // than erroring — a single defect never collapses the whole call to
803        // INTERNAL_ERROR.
804        let xml = r#"<html><body><p>Prefix kept here</p><p>Bad & entity halts</body>"#;
805        let res = parse_ar5iv(xml.as_bytes());
806        assert!(res.is_ok(), "hard syntax error must NOT error: {res:?}");
807        let (_title, sections) = res.expect("ok");
808        let joined: String = sections
809            .iter()
810            .map(|s| s.text.as_str())
811            .collect::<Vec<_>>()
812            .join(" ");
813        assert!(
814            joined.contains("Prefix kept here"),
815            "the prefix before the hard error is retained: {joined:?}"
816        );
817    }
818
819    // ---- apply_max_chars -------------------------------------------------
820
821    fn full_fixture() -> PaperText {
822        PaperText {
823            arxiv_id: "2401.12345".into(),
824            source: TextSource::Ar5iv,
825            title: Some("T".into()),
826            sections: vec![
827                TextSection {
828                    heading: None,
829                    text: "abcde".into(),
830                }, // 5
831                TextSection {
832                    heading: Some("H".into()),
833                    text: "fghij".into(),
834                }, // 5
835            ],
836            char_count: 10,
837            truncated: false,
838            retrieved_from: "https://ar5iv.labs.arxiv.org/html/2401.12345".into(),
839        }
840    }
841
842    #[test]
843    fn max_chars_none_returns_full() {
844        let out = apply_max_chars(full_fixture(), None);
845        assert!(!out.truncated);
846        assert_eq!(out.char_count, 10);
847        assert_eq!(out.sections.len(), 2);
848    }
849
850    #[test]
851    fn max_chars_above_total_is_untruncated() {
852        let out = apply_max_chars(full_fixture(), Some(100));
853        assert!(!out.truncated);
854        assert_eq!(out.char_count, 10);
855    }
856
857    #[test]
858    fn max_chars_cuts_within_a_section() {
859        // 7 chars: first section (5) whole, second cut to 2 ("fg").
860        let out = apply_max_chars(full_fixture(), Some(7));
861        assert!(out.truncated);
862        assert_eq!(out.char_count, 7);
863        assert_eq!(out.sections.len(), 2);
864        assert_eq!(out.sections[1].text, "fg");
865        assert_eq!(out.sections[1].heading.as_deref(), Some("H"));
866    }
867
868    #[test]
869    fn max_chars_drops_trailing_sections_on_exact_boundary() {
870        // 5 chars: first section exactly fits; the second is dropped.
871        let out = apply_max_chars(full_fixture(), Some(5));
872        assert!(out.truncated);
873        assert_eq!(out.char_count, 5);
874        assert_eq!(out.sections.len(), 1);
875    }
876
877    #[test]
878    fn max_chars_zero_yields_no_text_but_flags_truncated() {
879        let out = apply_max_chars(full_fixture(), Some(0));
880        assert!(out.truncated);
881        assert_eq!(out.char_count, 0);
882        assert!(out.sections.is_empty());
883    }
884
885    #[test]
886    fn max_chars_truncation_is_char_boundary_safe_for_multibyte() {
887        // Truncation must cut on `char` boundaries — a naive byte slice of a
888        // multibyte string (CJK / emoji) would panic. 7 chars, multi-byte.
889        let full = PaperText {
890            arxiv_id: "2401.12345".into(),
891            source: TextSource::Ar5iv,
892            title: None,
893            sections: vec![TextSection {
894                heading: None,
895                text: "あいうえお漢字".into(),
896            }],
897            char_count: 7,
898            truncated: false,
899            retrieved_from: "u".into(),
900        };
901        let out = apply_max_chars(full, Some(3));
902        assert!(out.truncated);
903        assert_eq!(out.char_count, 3);
904        assert_eq!(out.sections[0].text, "あいう");
905    }
906
907    // ---- url builder -----------------------------------------------------
908
909    #[test]
910    fn ar5iv_url_new_and_old_style() {
911        let base = Url::parse(AR5IV_DEFAULT_BASE).expect("base");
912        let new = ar5iv_url(&base, &ArxivId::parse("2401.12345").unwrap()).expect("url");
913        assert_eq!(new.path(), "/html/2401.12345");
914        assert_eq!(new.host_str(), Some("ar5iv.labs.arxiv.org"));
915        let old = ar5iv_url(&base, &ArxivId::parse("cond-mat/9501001").unwrap()).expect("url");
916        assert_eq!(old.path(), "/html/cond-mat/9501001");
917    }
918
919    // ---- cache round-trip ------------------------------------------------
920
921    #[test]
922    fn cache_write_then_read_round_trips() {
923        let dir = TempDir::new().unwrap();
924        let root = Utf8Path::from_path(dir.path()).unwrap();
925        let id = ArxivId::parse("2401.12345").unwrap();
926        let now = Utc::now();
927        assert!(cache_write_at(root, &id, &full_fixture(), now));
928        let got = cache_read_at(root, &id, now).expect("cache hit");
929        assert_eq!(got.arxiv_id, "2401.12345");
930        assert_eq!(got.sections.len(), 2);
931        assert!(!got.truncated, "cache stores the full, untruncated text");
932    }
933
934    #[test]
935    fn cache_miss_when_expired() {
936        let dir = TempDir::new().unwrap();
937        let root = Utf8Path::from_path(dir.path()).unwrap();
938        let id = ArxivId::parse("2401.12345").unwrap();
939        let written = Utc::now();
940        assert!(cache_write_at(root, &id, &full_fixture(), written));
941        let later = written + Duration::days(TEXT_CACHE_TTL_DAYS + 1);
942        assert!(cache_read_at(root, &id, later).is_none());
943    }
944
945    #[test]
946    fn cache_file_path_uses_text_dir_and_safekey() {
947        let root = Utf8Path::new("/tmp/cache");
948        let id = ArxivId::parse("2401.12345").unwrap();
949        let p = cache_file(root, &id);
950        assert!(p.components().any(|c| c.as_str() == "text"));
951        assert!(p.as_str().ends_with(".json"));
952    }
953
954    // ---- paper_text end-to-end (wiremock) --------------------------------
955
956    #[tokio::test]
957    async fn paper_text_fetches_parses_and_logs() {
958        let server = MockServer::start().await;
959        Mock::given(method("GET"))
960            .and(path_matcher("/html/2401.12345"))
961            .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
962            .mount(&server)
963            .await;
964
965        let host = server
966            .uri()
967            .parse::<Url>()
968            .unwrap()
969            .host_str()
970            .unwrap()
971            .to_string();
972        let (_td, ctx) = build_test_context(&host, None);
973        let log_path = ctx.log.path().to_path_buf();
974        let base = Url::parse(&server.uri()).expect("wiremock URI parses");
975        let id = ArxivId::parse("2401.12345").unwrap();
976
977        let out = paper_text(&base, &id, None, &ctx).await.expect("ok");
978        assert_eq!(out.arxiv_id, "2401.12345");
979        assert_eq!(out.source, TextSource::Ar5iv);
980        assert_eq!(out.title.as_deref(), Some("Tropical Tensor Networks"));
981        assert_eq!(out.sections.len(), 3);
982        assert!(!out.truncated);
983
984        // Exactly one Fetch provenance row, attributed to the ar5iv source.
985        let rows = read_rows(&log_path);
986        assert_eq!(rows.len(), 1, "one fetch row expected");
987        assert_eq!(rows[0].source.as_deref(), Some("ar5iv"));
988        assert_eq!(rows[0].ref_.as_deref(), Some("2401.12345"));
989        assert!(rows[0].error_code.is_none());
990    }
991
992    #[tokio::test]
993    async fn paper_text_truncates_when_max_chars_set() {
994        let server = MockServer::start().await;
995        Mock::given(method("GET"))
996            .and(path_matcher("/html/2401.12345"))
997            .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
998            .mount(&server)
999            .await;
1000        let host = server
1001            .uri()
1002            .parse::<Url>()
1003            .unwrap()
1004            .host_str()
1005            .unwrap()
1006            .to_string();
1007        let (_td, ctx) = build_test_context(&host, None);
1008        let base = Url::parse(&server.uri()).expect("uri");
1009        let id = ArxivId::parse("2401.12345").unwrap();
1010
1011        let out = paper_text(&base, &id, Some(10), &ctx).await.expect("ok");
1012        assert!(out.truncated);
1013        assert_eq!(out.char_count, 10);
1014    }
1015
1016    #[tokio::test]
1017    async fn paper_text_second_call_is_served_from_cache() {
1018        // First call hits the (single-response) mock and populates the
1019        // cache; the mock is mounted `up_to_n_times(1)`, so a second
1020        // network call would fail — proving the second call is cached.
1021        let server = MockServer::start().await;
1022        Mock::given(method("GET"))
1023            .and(path_matcher("/html/2401.12345"))
1024            .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_AR5IV))
1025            .up_to_n_times(1)
1026            .mount(&server)
1027            .await;
1028
1029        let host = server
1030            .uri()
1031            .parse::<Url>()
1032            .unwrap()
1033            .host_str()
1034            .unwrap()
1035            .to_string();
1036        let cache_dir = TempDir::new().unwrap();
1037        let cache_root =
1038            Utf8PathBuf::try_from(cache_dir.path().to_path_buf()).expect("utf8 cache root");
1039        let (_td, ctx) = build_test_context(&host, Some(cache_root));
1040        let base = Url::parse(&server.uri()).expect("uri");
1041        let id = ArxivId::parse("2401.12345").unwrap();
1042
1043        let first = paper_text(&base, &id, None, &ctx).await.expect("first ok");
1044        assert_eq!(first.sections.len(), 3);
1045        // Second call: no network response available; must be a cache hit.
1046        let second = paper_text(&base, &id, None, &ctx)
1047            .await
1048            .expect("second call served from cache");
1049        assert_eq!(second.sections.len(), 3);
1050        assert_eq!(second.title, first.title);
1051    }
1052
1053    #[tokio::test]
1054    async fn paper_text_empty_body_is_text_unavailable() {
1055        let server = MockServer::start().await;
1056        Mock::given(method("GET"))
1057            .and(path_matcher("/html/2401.99999"))
1058            .respond_with(
1059                ResponseTemplate::new(200)
1060                    .set_body_string("<html><head></head><body></body></html>"),
1061            )
1062            .mount(&server)
1063            .await;
1064        let host = server
1065            .uri()
1066            .parse::<Url>()
1067            .unwrap()
1068            .host_str()
1069            .unwrap()
1070            .to_string();
1071        let (_td, ctx) = build_test_context(&host, None);
1072        let base = Url::parse(&server.uri()).expect("uri");
1073        let id = ArxivId::parse("2401.99999").unwrap();
1074
1075        // A 200 that parses to nothing is "this representation is missing",
1076        // NOT "the id does not exist" (issue #302): the arXiv id was already
1077        // validated, so it must surface as TextUnavailable / TEXT_UNAVAILABLE
1078        // so an agent fetches the PDF rather than treating the ref as wrong.
1079        let err = paper_text(&base, &id, None, &ctx)
1080            .await
1081            .expect_err("empty body must be TextUnavailable");
1082        assert!(
1083            matches!(err, FetchError::TextUnavailable { .. }),
1084            "got {err:?}"
1085        );
1086        assert_eq!(
1087            crate::ErrorCode::from(&err),
1088            crate::ErrorCode::TextUnavailable
1089        );
1090    }
1091
1092    #[tokio::test]
1093    async fn paper_text_prose_free_body_is_text_unavailable() {
1094        // Issue #302's actual repro: a 200 whose body parses to a NON-empty
1095        // `sections` vec (a heading shell) but zero extractable prose
1096        // (`char_count == 0`). The old gate keyed on `sections.is_empty()`,
1097        // so this slipped through as an empty `Ok` → a silent exit-0 the
1098        // caller misread as a bad identifier. The `char_count == 0` gate
1099        // catches it.
1100        let server = MockServer::start().await;
1101        Mock::given(method("GET"))
1102            .and(path_matcher("/html/2012.03644"))
1103            .respond_with(
1104                ResponseTemplate::new(200).set_body_string(
1105                    "<html><head></head><body><h2>1 Introduction</h2></body></html>",
1106                ),
1107            )
1108            .mount(&server)
1109            .await;
1110        let host = server
1111            .uri()
1112            .parse::<Url>()
1113            .unwrap()
1114            .host_str()
1115            .unwrap()
1116            .to_string();
1117        let (_td, ctx) = build_test_context(&host, None);
1118        let base = Url::parse(&server.uri()).expect("uri");
1119        let id = ArxivId::parse("2012.03644").unwrap();
1120
1121        let err = paper_text(&base, &id, None, &ctx)
1122            .await
1123            .expect_err("a body with headings but no prose must be TextUnavailable");
1124        assert!(
1125            matches!(err, FetchError::TextUnavailable { .. }),
1126            "got {err:?}"
1127        );
1128    }
1129
1130    #[tokio::test]
1131    async fn paper_text_404_surfaces_http_error() {
1132        let server = MockServer::start().await;
1133        Mock::given(method("GET"))
1134            .and(path_matcher("/html/2401.00000"))
1135            .respond_with(ResponseTemplate::new(404))
1136            .mount(&server)
1137            .await;
1138        let host = server
1139            .uri()
1140            .parse::<Url>()
1141            .unwrap()
1142            .host_str()
1143            .unwrap()
1144            .to_string();
1145        let (_td, ctx) = build_test_context(&host, None);
1146        let base = Url::parse(&server.uri()).expect("uri");
1147        let id = ArxivId::parse("2401.00000").unwrap();
1148
1149        let err = paper_text(&base, &id, None, &ctx)
1150            .await
1151            .expect_err("404 must surface");
1152        // 404 collapses to NOT_FOUND at the boundary.
1153        assert_eq!(crate::ErrorCode::from(&err), crate::ErrorCode::NotFound);
1154    }
1155
1156    #[test]
1157    fn text_source_serializes_lowercase() {
1158        let s = serde_json::to_string(&TextSource::Ar5iv).expect("serialize");
1159        assert_eq!(s, "\"ar5iv\"");
1160    }
1161}