Skip to main content

doiget_core/
paper_tex_source.rs

1//! Fetch the raw **LaTeX source** of an arXiv paper from the arXiv source API
2//! (`https://export.arxiv.org/src/<id>`).
3//!
4//! This is the complement to [`crate::paper_text`] (ar5iv HTML extraction):
5//! ar5iv renders only papers that have been through the LaTeXML pipeline; the
6//! source API is always available for arXiv submissions that have a TeX source
7//! (PDF-only submissions return `NO_OA_AVAILABLE`).
8//!
9//! ## Why TeX source instead of ar5iv HTML?
10//!
11//! ar5iv HTML extraction (`paper_text`) is best-effort and unavailable for
12//! papers that were never processed by LaTeXML. The raw TeX source — when the
13//! submission has one — is the authoritative structured text. LLMs handle
14//! LaTeX well: `\section{}`, `\begin{equation}…\end{equation}`, etc. provide
15//! explicit structure that is often more reliable than ar5iv's HTML rendering.
16//!
17//! ## Source (arXiv E-print API)
18//!
19//! arXiv serves submission sources at
20//! `https://export.arxiv.org/src/<arxiv_id>`. The response is:
21//!
22//! - **Gzip'd tar** for multi-file submissions.
23//! - **Gzip'd single file** for single-file submissions.
24//! - **Raw PDF bytes** (`%PDF-` magic) for PDF-only submissions — no TeX
25//!   source available; yields `TextUnavailable`.
26//!
27//! Detection is by magic bytes on the response body.
28//!
29//! ## Source key
30//!
31//! Uses the existing `"arxiv"` HTTP source key (registered in
32//! [`crate::http::tier_1_allowlist`]), which covers `export.arxiv.org`.
33//! The provenance `source` field is labelled `"arxiv-src"` to distinguish
34//! TeX-source fetches from PDF fetches in the audit trail.
35//!
36//! ## Capability tier
37//!
38//! Tier 1 OA metadata, **always-on**: no env gate, no Cargo feature gate.
39//! Read-only, open-access, same posture class as [`crate::paper_text`]
40//! (ADR-0032 D2). TeX source is a structured text artifact, never a PDF
41//! reinterpretation (ADR-0032 D1 carve).
42//!
43//! ## Caching
44//!
45//! Results are cached at `<cache_root>/tex-src/<safekey>.json`. Best-effort:
46//! cache failures degrade to a re-fetch, never an error.
47
48use camino::{Utf8Path, Utf8PathBuf};
49use chrono::{DateTime, Duration, Utc};
50use flate2::read::GzDecoder;
51use serde::{Deserialize, Serialize};
52use std::io::Read;
53use tar::Archive;
54use url::Url;
55
56use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
57use crate::source::{FetchContext, FetchError};
58use crate::{ArxivId, Ref};
59
60/// HTTP-client source key. Reuses `"arxiv"` (covers `export.arxiv.org`).
61const HTTP_SOURCE_KEY: &str = "arxiv";
62
63/// Provenance audit label for TeX-source fetches.
64const PROV_SOURCE_LABEL: &str = "arxiv-src";
65
66/// Provenance audit label for source-bundle / figure fetches (ADR-0034 I4),
67/// distinct from [`PROV_SOURCE_LABEL`] so the audit log tells a `tex-source`
68/// text fetch apart from a `source` bundle/figures fetch.
69const PROV_SOURCE_BUNDLE_LABEL: &str = "arxiv-src-bundle";
70
71/// Production arXiv source API base. Overridable via
72/// `DOIGET_ARXIV_SRC_BASE` for tests.
73pub const ARXIV_SRC_DEFAULT_BASE: &str = "https://export.arxiv.org";
74
75// arXiv sources can be revised (v2, v3 can appear within days of v1); 7 days
76// balances freshness against re-fetch cost for stable papers.
77const TEX_SRC_CACHE_TTL_DAYS: i64 = 7;
78const TEX_SRC_CACHE_SCHEMA_VERSION: &str = "1.0";
79
80#[derive(Debug, Serialize, Deserialize)]
81struct CacheEntry {
82    schema_version: String,
83    /// RFC 3339 timestamp; matches `fetched_at` in `TextCacheEntry` and
84    /// `resolver_cache::CacheEntry` for consistency across cache formats.
85    fetched_at: String,
86    /// Stored explicitly so future versions can adjust per-entry TTL on read
87    /// without a code change (matches the pattern in `resolver_cache`).
88    ttl_seconds: i64,
89    inner: PaperTexSource,
90}
91
92/// Typed result from [`extract_tex`].
93///
94/// Using a named struct prevents accidental `(content, main_file)` swap bugs
95/// at the destructuring site (both members are `String` / `Option<String>`
96/// and would compile silently if swapped).
97#[derive(Debug)]
98pub(crate) struct ExtractedTex {
99    /// Filename of the main `.tex` file within the source tarball.
100    /// `None` when the submission was a single gzip'd file (no tar wrapper).
101    pub main_file: Option<String>,
102    /// Raw LaTeX content of the selected file.
103    pub content: String,
104}
105
106/// The raw LaTeX source of an arXiv paper.
107#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
108pub struct PaperTexSource {
109    /// The arXiv id this source belongs to.
110    pub arxiv_id: String,
111    /// Filename of the main `.tex` file within the source tarball.
112    /// `None` when the submission was a single gzip'd file (no tar).
113    pub main_file: Option<String>,
114    /// Raw LaTeX source of the main `.tex` file.
115    pub tex_source: String,
116    /// `char`s in `tex_source` (after any truncation).
117    pub char_count: usize,
118    /// `true` when `max_chars` truncated the source.
119    pub truncated: bool,
120    /// Final URL the source was retrieved from (after redirects).
121    pub retrieved_from: String,
122}
123
124/// Fetch the LaTeX source of an arXiv paper.
125///
126/// `base` is the arXiv export base URL (production [`ARXIV_SRC_DEFAULT_BASE`];
127/// tests inject a wiremock origin). `max_chars` caps the returned
128/// `tex_source` character count (`None` = no cap); truncation is flagged,
129/// never silent.
130///
131/// # Errors
132///
133/// - [`FetchError::Http`] — transport / status failure.
134/// - [`FetchError::TextUnavailable`] — PDF-only submission or no `.tex` files
135///   in the tarball.
136/// - [`FetchError::SourceSchema`] — URL construction or gzip/tar parse error.
137/// - [`FetchError::Log`] — provenance append failed (fail-closed).
138pub async fn paper_tex_source(
139    base: &Url,
140    id: &ArxivId,
141    max_chars: Option<usize>,
142    ctx: &FetchContext,
143) -> Result<PaperTexSource, FetchError> {
144    if let Some(root) = &ctx.cache_root {
145        if let Some(full) = cache_read(root, id) {
146            return Ok(apply_max_chars(full, max_chars));
147        }
148    }
149
150    let full = fetch_and_extract(base, id, ctx).await?;
151
152    if let Some(root) = &ctx.cache_root {
153        if !cache_write(root, id, &full) {
154            tracing::warn!(
155                cache_root = %root,
156                arxiv_id = %id.as_str(),
157                "tex-source cache write failed; next request will re-fetch"
158            );
159        }
160    }
161
162    Ok(apply_max_chars(full, max_chars))
163}
164
165async fn fetch_and_extract(
166    base: &Url,
167    id: &ArxivId,
168    ctx: &FetchContext,
169) -> Result<PaperTexSource, FetchError> {
170    let _permit = ctx.rate_limiter.acquire(HTTP_SOURCE_KEY).await;
171
172    let url = src_url(base, id)?;
173    let (body, final_url) = ctx.http.fetch_bytes(HTTP_SOURCE_KEY, url).await?;
174
175    let extracted = extract_tex(id, &body)?;
176    let char_count = extracted.content.chars().count();
177
178    let canonical = Ref::Arxiv(id.clone())
179        .promote(PROV_SOURCE_LABEL, None)
180        .digest_hex();
181    ctx.log.append(RowInput {
182        event: LogEvent::Fetch,
183        result: LogResult::Ok,
184        capability: Capability::Oa,
185        ref_: Some(id.as_str()),
186        source: Some(PROV_SOURCE_LABEL),
187        error_code: None,
188        size_bytes: Some(body.len() as u64),
189        license: Some("arxiv-default"),
190        store_path: None,
191        canonical_digest: Some(&canonical),
192    })?;
193
194    Ok(PaperTexSource {
195        arxiv_id: id.as_str().to_string(),
196        main_file: extracted.main_file,
197        tex_source: extracted.content,
198        char_count,
199        truncated: false,
200        retrieved_from: final_url.to_string(),
201    })
202}
203
204fn src_url(base: &Url, id: &ArxivId) -> Result<Url, FetchError> {
205    base.join(&format!("/src/{}", id.as_str()))
206        .map_err(|e| FetchError::SourceSchema {
207            hint: format!("arXiv src URL construction failed: {e}"),
208        })
209}
210
211/// The shape of a decompressed arXiv `/src/<id>` response body, classified by
212/// magic bytes. Shared by the text path ([`extract_tex`]) and the bundle path
213/// ([`extract_bundle`]) so the gzip + ustar detection lives in one place
214/// (issue #346); each caller maps the variants to its own result type.
215#[derive(Debug)]
216enum SrcPayload {
217    /// `%PDF-` magic — a PDF-only submission (no source).
218    PdfOnly,
219    /// A single file: a bare uncompressed body, or a single gzip'd non-tar
220    /// file. The bytes are that file's content.
221    SingleFile(Vec<u8>),
222    /// A gzip'd `ustar` tar archive; the bytes are the decompressed tar.
223    Tar(Vec<u8>),
224}
225
226/// Classify + decompress an arXiv `/src/` body by magic bytes.
227///
228/// `max_decompressed` caps the gzip OUTPUT size when `Some` (the bundle path,
229/// against a gzip bomb — ADR-0034 I5); `None` leaves the text path's
230/// decompression byte-identical to the pre-refactor inline form (ADR-0034 D6).
231///
232/// # Errors
233///
234/// [`FetchError::SourceSchema`] on a gzip decode failure or an over-cap body.
235fn classify_src(bytes: &[u8], max_decompressed: Option<u64>) -> Result<SrcPayload, FetchError> {
236    if bytes.starts_with(b"%PDF-") {
237        return Ok(SrcPayload::PdfOnly);
238    }
239    // Not gzip (magic `1f 8b`) → a bare uncompressed single file (no tar).
240    if bytes.len() < 2 || bytes[0..2] != [0x1f, 0x8b] {
241        return Ok(SrcPayload::SingleFile(bytes.to_vec()));
242    }
243    let mut decompressed = Vec::new();
244    match max_decompressed {
245        Some(cap) => {
246            // `take(cap + 1)` bounds the decompressed bytes; a result longer
247            // than `cap` means the (capped) stream was truncated → reject.
248            let mut gz = GzDecoder::new(std::io::Cursor::new(bytes)).take(cap + 1);
249            gz.read_to_end(&mut decompressed)
250                .map_err(|e| FetchError::SourceSchema {
251                    hint: format!("gzip decompress of arXiv src failed: {e}"),
252                })?;
253            if decompressed.len() as u64 > cap {
254                return Err(FetchError::SourceSchema {
255                    hint: format!(
256                        "arXiv src decompressed size exceeds {cap} bytes \
257                         (possible gzip bomb); refusing"
258                    ),
259                });
260            }
261        }
262        None => {
263            let mut gz = GzDecoder::new(std::io::Cursor::new(bytes));
264            gz.read_to_end(&mut decompressed)
265                .map_err(|e| FetchError::SourceSchema {
266                    hint: format!("gzip decompress of arXiv src failed: {e}"),
267                })?;
268        }
269    }
270    // UStar tar detection: POSIX.1-1988 tar header magic at byte offset 257.
271    // A valid tar header is ≥ 512 bytes; the `> 262` guard is conservative
272    // (only 262 bytes are needed for the magic slice) and avoids a panic.
273    let is_tar = decompressed.len() > 262 && &decompressed[257..262] == b"ustar";
274    if is_tar {
275        Ok(SrcPayload::Tar(decompressed))
276    } else {
277        Ok(SrcPayload::SingleFile(decompressed))
278    }
279}
280
281/// Detect content type by magic bytes and extract the main LaTeX source.
282///
283/// Returns an [`ExtractedTex`] with `main_file` and `content`.
284pub(crate) fn extract_tex(id: &ArxivId, bytes: &[u8]) -> Result<ExtractedTex, FetchError> {
285    // Cap decompression against a gzip bomb (review #352): the HTTP layer only
286    // bounds the *compressed* body, so an unbounded `read_to_end` here could
287    // OOM on a crafted `/src` payload — now reachable via the MCP
288    // `doiget_paper_tex_source` tool. Real arXiv sources are far below the cap,
289    // so this supersedes ADR-0034 D6's "byte-identical" note for pathological
290    // inputs only. The single-file arm covers both a bare uncompressed `.tex`
291    // (arXiv occasionally serves one for trivial submissions) and a single
292    // gzip'd `.tex`.
293    match classify_src(bytes, Some(SRC_MAX_DECOMPRESSED_BYTES))? {
294        SrcPayload::PdfOnly => Err(FetchError::TextUnavailable {
295            arxiv_id: id.clone(),
296        }),
297        SrcPayload::SingleFile(data) => {
298            let text = String::from_utf8_lossy(&data).into_owned();
299            if text.trim().is_empty() {
300                return Err(FetchError::TextUnavailable {
301                    arxiv_id: id.clone(),
302                });
303            }
304            Ok(ExtractedTex {
305                main_file: None,
306                content: text,
307            })
308        }
309        SrcPayload::Tar(decompressed) => extract_from_tar(id, &decompressed),
310    }
311}
312
313/// Extract the main `.tex` file from an uncompressed tar archive using a
314/// weighted scoring heuristic:
315///
316///   score = (1 if `\documentclass` present) × 1_000_000
317///         + (1 if filename ends with `main.tex`) × 100_000
318///         + byte_count_of_file
319///
320/// The weights dominate realistic file sizes: a `\documentclass` file beats a
321/// non-`\documentclass` one unless the latter is ~1 MB larger, and within
322/// `\documentclass` files `main.tex` wins unless a rival is ~100 KB larger —
323/// neither happens for real sub-files. The sum uses `saturating_add` so it
324/// stays total-order-safe even for a pathological size the decompression cap
325/// would already reject (the previous note claiming "~1 GB overflows i64" was
326/// wrong — `i64::MAX` is ~9.2 EB; review #352).
327fn extract_from_tar(id: &ArxivId, bytes: &[u8]) -> Result<ExtractedTex, FetchError> {
328    let mut archive = Archive::new(std::io::Cursor::new(bytes));
329    let entries = archive.entries().map_err(|e| FetchError::SourceSchema {
330        hint: format!("tar read failed: {e}"),
331    })?;
332
333    let mut tex_files: Vec<(String, String)> = Vec::new();
334    // Track .tex entries attempted (even if read failed) so that a corrupt
335    // archive is distinguishable from a PDF-only submission.
336    let mut tex_attempted: usize = 0;
337    // Entries skipped because the header/path could not be parsed, the path was
338    // unsafe, or the body failed to read. Logged below so a partial extraction
339    // is never silent (mirrors `extract_bundle`'s discipline; review #352).
340    let mut unreadable: usize = 0;
341    for entry in entries {
342        let Ok(mut entry) = entry else {
343            unreadable += 1;
344            continue;
345        };
346        let raw = match entry.path() {
347            Ok(p) => p.to_string_lossy().to_string(),
348            Err(_) => {
349                unreadable += 1;
350                continue;
351            }
352        };
353        // Use the sanitised relative path for `main_file`: the text path never
354        // writes files, but the name flows into the CLI output / MCP envelope,
355        // so a crafted `../`-style entry name must never be surfaced to a
356        // caller (review #352).
357        let Some(path) = sanitize_entry_path(&raw).map(|p| p.to_string()) else {
358            tracing::warn!(arxiv_id = %id.as_str(), entry = %raw, "skipping unsafe arXiv src entry path");
359            continue;
360        };
361        if !path.ends_with(".tex") {
362            continue;
363        }
364        tex_attempted += 1;
365        let mut content = String::new();
366        match entry.read_to_string(&mut content) {
367            Ok(_) if !content.trim().is_empty() => tex_files.push((path, content)),
368            Ok(_) => {} // empty .tex — legitimately skipped, not a failure
369            Err(_) => unreadable += 1,
370        }
371    }
372    if unreadable > 0 {
373        tracing::warn!(
374            arxiv_id = %id.as_str(),
375            unreadable,
376            "some arXiv src tar entries were unreadable/unsafe and were skipped"
377        );
378    }
379
380    if tex_files.is_empty() {
381        // Distinguish "PDF-only" from "corrupt archive": if .tex entries were
382        // present but none could be read, this is a schema/decode error, not a
383        // missing-source condition (which would mislead agents into thinking
384        // the paper has no TeX source).
385        return Err(if tex_attempted > 0 {
386            FetchError::SourceSchema {
387                hint: format!("tar contained {tex_attempted} .tex entries but all failed to read"),
388            }
389        } else {
390            FetchError::TextUnavailable {
391                arxiv_id: id.clone(),
392            }
393        });
394    }
395
396    let best = tex_files.into_iter().max_by_key(|(name, content)| {
397        let docclass = i64::from(content.contains(r"\documentclass")) * 1_000_000;
398        let is_main = i64::from(name.ends_with("main.tex") || name == "main.tex") * 100_000;
399        let size = i64::try_from(content.len()).unwrap_or(i64::MAX);
400        docclass.saturating_add(is_main).saturating_add(size)
401    });
402
403    match best {
404        Some((name, content)) => Ok(ExtractedTex {
405            main_file: Some(name),
406            content,
407        }),
408        None => Err(FetchError::TextUnavailable {
409            arxiv_id: id.clone(),
410        }),
411    }
412}
413
414fn apply_max_chars(mut full: PaperTexSource, max_chars: Option<usize>) -> PaperTexSource {
415    let Some(max) = max_chars else {
416        return full;
417    };
418    if full.char_count <= max {
419        return full;
420    }
421    full.tex_source = full.tex_source.chars().take(max).collect();
422    full.char_count = max;
423    full.truncated = true;
424    full
425}
426
427fn cache_file(cache_root: &Utf8Path, id: &ArxivId) -> Utf8PathBuf {
428    let safekey = Ref::Arxiv(id.clone()).safekey();
429    cache_root
430        .join("tex-src")
431        .join(format!("{}.json", safekey.as_str()))
432}
433
434fn cache_read(cache_root: &Utf8Path, id: &ArxivId) -> Option<PaperTexSource> {
435    cache_read_at(cache_root, id, Utc::now())
436}
437
438fn cache_read_at(
439    cache_root: &Utf8Path,
440    id: &ArxivId,
441    now: DateTime<Utc>,
442) -> Option<PaperTexSource> {
443    let path = cache_file(cache_root, id);
444    let bytes = std::fs::read(&path).ok()?;
445    let entry: CacheEntry = serde_json::from_slice(&bytes).ok()?;
446    if entry.schema_version != TEX_SRC_CACHE_SCHEMA_VERSION {
447        return None;
448    }
449    let fetched = DateTime::parse_from_rfc3339(&entry.fetched_at)
450        .ok()?
451        .with_timezone(&Utc);
452    if now.signed_duration_since(fetched) > Duration::seconds(entry.ttl_seconds) {
453        return None;
454    }
455    Some(entry.inner)
456}
457
458fn cache_write(cache_root: &Utf8Path, id: &ArxivId, full: &PaperTexSource) -> bool {
459    cache_write_at(cache_root, id, full, Utc::now())
460}
461
462fn cache_write_at(
463    cache_root: &Utf8Path,
464    id: &ArxivId,
465    full: &PaperTexSource,
466    now: DateTime<Utc>,
467) -> bool {
468    let path = cache_file(cache_root, id);
469    if let Some(dir) = path.parent() {
470        if std::fs::create_dir_all(dir).is_err() {
471            return false;
472        }
473    }
474    let entry = CacheEntry {
475        schema_version: TEX_SRC_CACHE_SCHEMA_VERSION.to_string(),
476        fetched_at: now.to_rfc3339(),
477        ttl_seconds: TEX_SRC_CACHE_TTL_DAYS * 86_400,
478        inner: full.clone(),
479    };
480    match serde_json::to_vec(&entry) {
481        Ok(bytes) => std::fs::write(&path, bytes).is_ok(),
482        Err(_) => false,
483    }
484}
485
486/// Resolve the arXiv source base URL.
487pub fn resolve_arxiv_src_base() -> Result<Url, String> {
488    let raw = std::env::var("DOIGET_ARXIV_SRC_BASE")
489        .unwrap_or_else(|_| ARXIV_SRC_DEFAULT_BASE.to_string());
490    Url::parse(&raw).map_err(|e| format!("DOIGET_ARXIV_SRC_BASE is not a valid URL: {e}"))
491}
492
493// ─────────────────────────────────────────────────────────────────────────────
494// Source bundle / figures (ADR-0034). The arXiv `/src/<id>` tarball already
495// downloaded for the text path carries EVERY submission file; this section
496// surfaces the full bundle (or figures only) instead of discarding them. Every
497// returned path is sanitised (relative, no `..`, no anchor) so a caller can
498// join it under any output directory without escaping it (zip-slip, ADR-0034 D3).
499// ─────────────────────────────────────────────────────────────────────────────
500
501/// Image/figure file extensions (lowercase, no dot). Saved opaque — never
502/// interpreted (ADR-0034 D2). Vector `.pdf` figures are included.
503const FIGURE_EXTS: &[&str] = &["pdf", "eps", "ps", "png", "jpg", "jpeg", "gif", "svg"];
504
505/// Cap on the DECOMPRESSED size of an arXiv `/src/` tarball (ADR-0034 I5).
506/// The HTTP client already caps the *compressed* download at `PDF_MAX_BYTES`
507/// (100 MB), but a small gzip can expand to many GB; this bounds the
508/// decompressed bytes held in memory to refuse a gzip bomb. Generous vs real
509/// arXiv sources (rarely > 100 MB decompressed), strict vs a multi-GB bomb.
510const SRC_MAX_DECOMPRESSED_BYTES: u64 = 500_000_000;
511
512/// Which subset of the source tarball to materialise.
513#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
514pub enum BundleFilter {
515    /// Every regular file in the tarball.
516    All,
517    /// Only image/figure files (by the `FIGURE_EXTS` extension allowlist).
518    FiguresOnly,
519}
520
521/// One file extracted from an arXiv source tarball.
522#[derive(Debug, Clone, PartialEq, Eq)]
523#[non_exhaustive]
524pub struct SourceFile {
525    /// Sanitised **relative** path (never absolute, never contains `..`),
526    /// safe to join under any output root (ADR-0034 D3). The field is
527    /// `pub(crate)` so it can only be set by `extract_bundle` — which runs
528    /// every path through `sanitize_entry_path` — and an external caller
529    /// cannot forge a `SourceFile` carrying an unsafe path. This mirrors the
530    /// checked-construction pattern of `Doi` / `ArxivId`. Read it via
531    /// [`SourceFile::path`].
532    pub(crate) path: Utf8PathBuf,
533    /// Raw file bytes, opaque (never interpreted; ADR-0034 D2).
534    pub bytes: Vec<u8>,
535}
536
537impl SourceFile {
538    /// The sanitised relative path of this file (never absolute, no `..`).
539    #[must_use]
540    pub fn path(&self) -> &Utf8Path {
541        &self.path
542    }
543}
544
545/// Sanitise a raw tar entry path into a safe **relative** path, or `None` to
546/// reject it (zip-slip / path-traversal guard, ADR-0034 D3).
547///
548/// Rejects: absolute / root-anchored paths (leading `/` or `\`, or a Windows
549/// drive prefix like `C:`); any `..` component; any component containing `:`
550/// or a NUL byte; and paths with no normal component. Splits on BOTH `/` and
551/// `\` so a Windows-style traversal in a Unix-produced tar is caught
552/// regardless of the extracting platform. The result is always relative with
553/// no `..`, so `root.join(result)` cannot escape `root`.
554fn sanitize_entry_path(raw: &str) -> Option<Utf8PathBuf> {
555    if raw.is_empty() || raw.contains('\0') {
556        return None;
557    }
558    // Absolute / root-anchored — anomalous in an arXiv source tarball.
559    if raw.starts_with('/') || raw.starts_with('\\') {
560        return None;
561    }
562    let b = raw.as_bytes();
563    // Windows drive prefix `X:` / `X:\`.
564    if b.len() >= 2 && b[0].is_ascii_alphabetic() && b[1] == b':' {
565        return None;
566    }
567    let mut out = Utf8PathBuf::new();
568    let mut any = false;
569    for seg in raw.split(['/', '\\']) {
570        match seg {
571            "" | "." => continue, // collapse `//`, drop `.`
572            ".." => return None,  // traversal — reject the whole path
573            s => {
574                if s.contains(':') || s.contains('\0') {
575                    return None;
576                }
577                out.push(s);
578                any = true;
579            }
580        }
581    }
582    if any {
583        Some(out)
584    } else {
585        None
586    }
587}
588
589/// True when `path`'s extension is in the figure allowlist (case-insensitive).
590fn is_figure(path: &Utf8Path) -> bool {
591    match path.extension() {
592        Some(ext) => FIGURE_EXTS.contains(&ext.to_ascii_lowercase().as_str()),
593        None => false,
594    }
595}
596
597/// Decompress + untar an arXiv `/src/` body and collect the selected files.
598///
599/// Applies the same PDF / gzip / ustar magic-byte checks as [`extract_tex`],
600/// but a PDF-only response, a bare uncompressed file, or a single gzip'd file
601/// yields [`FetchError::SourceUnavailable`] — there is no multi-file bundle in
602/// a single-file response (unlike the text path, which passes a bare `.tex`
603/// through; the existing `extract_tex` text path is left byte-identical,
604/// ADR-0034 D6). Decompression is size-capped against a gzip bomb (ADR-0034
605/// I5). Only **regular** tar entries are considered — symlinks / hardlinks /
606/// devices are skipped (a symlink is itself a traversal vector, ADR-0034 D3).
607/// Every path is run through [`sanitize_entry_path`]; a path-rejected entry is
608/// skipped with a `tracing::warn!`. Entries that fail to read (malformed
609/// header, non-decodable path, or `read_to_end` error) are skipped, logged,
610/// and counted, so an empty result distinguishes a corrupt archive
611/// ([`FetchError::SourceSchema`]) from genuinely no matching files
612/// ([`FetchError::SourceUnavailable`]) (ADR-0034 C1).
613pub(crate) fn extract_bundle(
614    id: &ArxivId,
615    bytes: &[u8],
616    filter: BundleFilter,
617) -> Result<Vec<SourceFile>, FetchError> {
618    // PDF-only / bare single file / single gzip'd file: no multi-file bundle.
619    // Decompression is size-capped against a gzip bomb (ADR-0034 I5); the
620    // shared [`classify_src`] keeps the gzip/ustar detection in one place (#346).
621    let decompressed = match classify_src(bytes, Some(SRC_MAX_DECOMPRESSED_BYTES))? {
622        SrcPayload::Tar(d) => d,
623        SrcPayload::PdfOnly | SrcPayload::SingleFile(_) => return Err(no_files(id, filter)),
624    };
625
626    let mut archive = Archive::new(std::io::Cursor::new(decompressed));
627    let entries = archive.entries().map_err(|e| FetchError::SourceSchema {
628        hint: format!("tar read failed: {e}"),
629    })?;
630
631    let mut files: Vec<SourceFile> = Vec::new();
632    // Count entries we matched but could not materialise, so an empty result
633    // distinguishes a corrupt/unreadable archive (SourceSchema) from a
634    // genuinely absent bundle (SourceUnavailable) — mirrors
635    // extract_from_tar's tex_attempted (ADR-0034 C1). Path-rejected
636    // (zip-slip) and filtered-out entries are deliberate skips, NOT counted.
637    let mut unreadable: usize = 0;
638    for entry in entries {
639        let mut entry = match entry {
640            Ok(e) => e,
641            Err(e) => {
642                unreadable += 1;
643                tracing::warn!(arxiv_id = %id.as_str(), error = %e, "arXiv src: skipping malformed tar entry");
644                continue;
645            }
646        };
647        // Regular files only: a symlink/hardlink entry is a traversal vector
648        // and is never needed for source/figures (ADR-0034 D3).
649        if !entry.header().entry_type().is_file() {
650            continue;
651        }
652        let raw_path = match entry.path() {
653            Ok(p) => p.to_string_lossy().into_owned(),
654            Err(e) => {
655                unreadable += 1;
656                tracing::warn!(arxiv_id = %id.as_str(), error = %e, "arXiv src: tar entry has a non-decodable path; skipping");
657                continue;
658            }
659        };
660        let Some(safe) = sanitize_entry_path(&raw_path) else {
661            tracing::warn!(
662                entry = %raw_path,
663                "arXiv src: rejected unsafe tar entry path (zip-slip guard)"
664            );
665            continue;
666        };
667        if filter == BundleFilter::FiguresOnly && !is_figure(&safe) {
668            continue;
669        }
670        let mut buf = Vec::new();
671        match entry.read_to_end(&mut buf) {
672            Ok(_) => files.push(SourceFile {
673                path: safe,
674                bytes: buf,
675            }),
676            Err(e) => {
677                unreadable += 1;
678                tracing::warn!(arxiv_id = %id.as_str(), entry = %safe, error = %e, "arXiv src: failed to read tar entry; skipping");
679            }
680        }
681    }
682
683    if files.is_empty() {
684        // Corrupt/unreadable archive vs genuinely no matching files (ADR-0034 C1).
685        return Err(if unreadable > 0 {
686            FetchError::SourceSchema {
687                hint: format!(
688                    "arXiv src tar had {unreadable} unreadable entr(y/ies) and no usable files"
689                ),
690            }
691        } else {
692            no_files(id, filter)
693        });
694    }
695    if unreadable > 0 {
696        tracing::warn!(
697            arxiv_id = %id.as_str(),
698            unreadable,
699            extracted = files.len(),
700            "arXiv src: bundle is partial — some entries were unreadable and skipped"
701        );
702    }
703    Ok(files)
704}
705
706/// The "no usable files" error for a `source` fetch, labelled with the
707/// requested representation so the message is accurate (not ar5iv-specific;
708/// ADR-0034 I2).
709fn no_files(id: &ArxivId, filter: BundleFilter) -> FetchError {
710    FetchError::SourceUnavailable {
711        arxiv_id: id.clone(),
712        kind: match filter {
713            BundleFilter::All => "source bundle",
714            BundleFilter::FiguresOnly => "figures",
715        },
716    }
717}
718
719/// Fetch the arXiv source bundle (or figures only) for `id`.
720///
721/// Tier-1 OA, always-on (ADR-0034 D1). Performs the SAME single `/src/<id>`
722/// request as [`paper_tex_source`] and returns the selected files **in
723/// memory**; the caller writes them to disk. Every returned path is sanitised
724/// (ADR-0034 D3). Not cached (ADR-0034 D5).
725///
726/// # Errors
727///
728/// - [`FetchError::Http`] — transport / status failure.
729/// - [`FetchError::TextUnavailable`] — PDF-only / single-file submission, or no
730///   matching files (e.g. `--figures-only` on a figure-less submission).
731/// - [`FetchError::SourceSchema`] — URL construction or gzip/tar parse error.
732/// - [`FetchError::Log`] — provenance append failed (fail-closed).
733pub async fn paper_source_bundle(
734    base: &Url,
735    id: &ArxivId,
736    filter: BundleFilter,
737    ctx: &FetchContext,
738) -> Result<Vec<SourceFile>, FetchError> {
739    let _permit = ctx.rate_limiter.acquire(HTTP_SOURCE_KEY).await;
740
741    let url = src_url(base, id)?;
742    let (body, _final_url) = ctx.http.fetch_bytes(HTTP_SOURCE_KEY, url).await?;
743
744    let files = extract_bundle(id, &body, filter)?;
745
746    let canonical = Ref::Arxiv(id.clone())
747        .promote(PROV_SOURCE_BUNDLE_LABEL, None)
748        .digest_hex();
749    ctx.log.append(RowInput {
750        event: LogEvent::Fetch,
751        result: LogResult::Ok,
752        capability: Capability::Oa,
753        ref_: Some(id.as_str()),
754        source: Some(PROV_SOURCE_BUNDLE_LABEL),
755        error_code: None,
756        size_bytes: Some(body.len() as u64),
757        license: Some("arxiv-default"),
758        store_path: None,
759        canonical_digest: Some(&canonical),
760    })?;
761
762    Ok(files)
763}
764
765#[cfg(test)]
766#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic, missing_docs)]
767mod tests {
768    use super::*;
769    use flate2::write::GzEncoder;
770    use flate2::Compression;
771    use std::io::Write as _;
772
773    fn make_id(s: &str) -> ArxivId {
774        match Ref::parse(s).expect("parse") {
775            Ref::Arxiv(a) => a,
776            _ => panic!("expected arxiv id"),
777        }
778    }
779
780    fn gzip_bytes(data: &[u8]) -> Vec<u8> {
781        let mut enc = GzEncoder::new(Vec::new(), Compression::default());
782        enc.write_all(data).expect("gzip write");
783        enc.finish().expect("gzip finish")
784    }
785
786    fn tar_gzip(files: &[(&str, &[u8])]) -> Vec<u8> {
787        let mut builder = tar::Builder::new(Vec::new());
788        for (name, data) in files {
789            let mut header = tar::Header::new_gnu();
790            header.set_size(data.len() as u64);
791            header.set_mode(0o644);
792            header.set_cksum();
793            builder
794                .append_data(&mut header, name, std::io::Cursor::new(data))
795                .expect("tar append");
796        }
797        gzip_bytes(&builder.into_inner().expect("tar finish"))
798    }
799
800    fn make_src(id: &ArxivId) -> PaperTexSource {
801        PaperTexSource {
802            arxiv_id: id.as_str().to_string(),
803            main_file: Some("main.tex".into()),
804            tex_source: "\\documentclass{article}".into(),
805            char_count: 23,
806            truncated: false,
807            retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
808        }
809    }
810
811    // ── apply_max_chars ───────────────────────────────────────────────────────
812
813    #[test]
814    fn apply_max_chars_no_cap_is_identity() {
815        let id = make_id("2401.12345");
816        let src = make_src(&id);
817        let out = apply_max_chars(src.clone(), None);
818        assert_eq!(out, src);
819    }
820
821    #[test]
822    fn apply_max_chars_truncates() {
823        let id = make_id("2401.12345");
824        let src = PaperTexSource {
825            arxiv_id: id.as_str().to_string(),
826            main_file: None,
827            tex_source: "abcdefghij".into(),
828            char_count: 10,
829            truncated: false,
830            retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
831        };
832        let out = apply_max_chars(src, Some(4));
833        assert_eq!(out.tex_source, "abcd");
834        assert_eq!(out.char_count, 4);
835        assert!(out.truncated);
836    }
837
838    // ── extract_tex: magic-byte paths ────────────────────────────────────────
839
840    #[test]
841    fn pdf_only_yields_text_unavailable() {
842        let id = make_id("2401.12345");
843        let result = extract_tex(&id, b"%PDF-1.4 fake");
844        assert!(matches!(result, Err(FetchError::TextUnavailable { .. })));
845    }
846
847    #[test]
848    fn raw_tex_passthrough() {
849        let id = make_id("2401.12345");
850        let tex = b"\\documentclass{article}\n\\begin{document}\nHello.\\end{document}";
851        let ext = extract_tex(&id, tex).expect("extract");
852        assert!(ext.main_file.is_none());
853        assert!(ext.content.contains("\\documentclass"));
854    }
855
856    #[test]
857    fn gzip_single_file_extracted() {
858        let id = make_id("2401.12345");
859        let tex = b"\\documentclass{article}\n\\begin{document}Hello\\end{document}";
860        let gz = gzip_bytes(tex);
861        let ext = extract_tex(&id, &gz).expect("extract");
862        assert!(ext.main_file.is_none(), "single gzip has no tar filename");
863        assert!(ext.content.contains("\\documentclass"));
864    }
865
866    // ── classify_src: gzip-bomb decompression cap (review #352) ───────────────
867
868    #[test]
869    fn classify_src_rejects_decompression_over_cap() {
870        // A body decompressing to more than the cap MUST be rejected
871        // (`SourceSchema`), never silently accepted — this is the gzip-bomb
872        // guard. Pins the wiring so a regression that drops the cap (e.g.
873        // passes `None` on the text path again) fails loudly. Uses a tiny cap
874        // so the test needs no large allocation.
875        let big = vec![b'x'; 10_000];
876        let gz = gzip_bytes(&big);
877        let err = classify_src(&gz, Some(1_000)).expect_err("over-cap must be rejected");
878        assert!(
879            matches!(err, FetchError::SourceSchema { .. }),
880            "got {err:?}"
881        );
882    }
883
884    #[test]
885    fn classify_src_accepts_decompression_within_cap() {
886        let small = vec![b'x'; 500];
887        let gz = gzip_bytes(&small);
888        let payload = classify_src(&gz, Some(1_000)).expect("within cap");
889        assert!(matches!(payload, SrcPayload::SingleFile(_)));
890    }
891
892    // ── extract_from_tar: selection heuristic ────────────────────────────────
893
894    #[test]
895    fn tar_selects_documentclass_file_over_plain() {
896        let id = make_id("2401.12345");
897        let payload = tar_gzip(&[
898            ("paper.tex", b"\\documentclass{article} main content"),
899            ("macros.tex", b"\\newcommand{\\foo}{bar}"),
900        ]);
901        let ext = extract_tex(&id, &payload).expect("extract");
902        assert_eq!(ext.main_file.as_deref(), Some("paper.tex"));
903        assert!(ext.content.contains("\\documentclass"));
904    }
905
906    #[test]
907    fn tar_prefers_main_tex_among_documentclass_files() {
908        let id = make_id("2401.12345");
909        let payload = tar_gzip(&[
910            ("other.tex", b"\\documentclass{article} other content here"),
911            ("main.tex", b"\\documentclass{article} main"),
912        ]);
913        let ext = extract_tex(&id, &payload).expect("extract");
914        assert_eq!(
915            ext.main_file.as_deref(),
916            Some("main.tex"),
917            "main.tex bonus must override smaller-but-also-documentclass other.tex"
918        );
919    }
920
921    #[test]
922    fn tar_falls_back_to_largest_file_when_no_documentclass() {
923        let id = make_id("2401.12345");
924        let short = b"\\section{Short}".as_slice();
925        let mut long_content = b"\\section{Long} ".to_vec();
926        long_content.extend(vec![b'x'; 500]);
927        let payload = tar_gzip(&[("short.tex", short), ("long.tex", &long_content)]);
928        let ext = extract_tex(&id, &payload).expect("extract");
929        assert_eq!(ext.main_file.as_deref(), Some("long.tex"));
930    }
931
932    #[test]
933    fn tar_with_no_tex_files_is_text_unavailable() {
934        let id = make_id("2401.12345");
935        let payload = tar_gzip(&[("README.md", b"# Paper"), ("figure.eps", b"%!PS")]);
936        let err = extract_tex(&id, &payload).expect_err("should fail");
937        assert!(matches!(err, FetchError::TextUnavailable { .. }));
938    }
939
940    // ── cache ────────────────────────────────────────────────────────────────
941
942    #[test]
943    fn resolve_base_defaults_to_production() {
944        if std::env::var("DOIGET_ARXIV_SRC_BASE").is_err() {
945            let u = resolve_arxiv_src_base().expect("resolve");
946            assert_eq!(u.as_str(), "https://export.arxiv.org/");
947        }
948    }
949
950    #[test]
951    fn cache_round_trip() {
952        let dir = tempfile::tempdir().expect("tempdir");
953        let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
954        let id = make_id("2401.12345");
955        let src = make_src(&id);
956        assert!(cache_write(&root, &id, &src));
957        let read = cache_read(&root, &id).expect("cache hit");
958        assert_eq!(read, src);
959    }
960
961    #[test]
962    fn cache_expired_returns_none() {
963        let dir = tempfile::tempdir().expect("tempdir");
964        let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
965        let id = make_id("2401.12345");
966        let src = PaperTexSource {
967            arxiv_id: id.as_str().to_string(),
968            main_file: None,
969            tex_source: "test".into(),
970            char_count: 4,
971            truncated: false,
972            retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
973        };
974        let past = Utc::now() - Duration::days(TEX_SRC_CACHE_TTL_DAYS + 1);
975        assert!(cache_write_at(&root, &id, &src, past));
976        assert!(cache_read_at(&root, &id, Utc::now()).is_none());
977    }
978
979    #[test]
980    fn cache_schema_version_mismatch_returns_none() {
981        let dir = tempfile::tempdir().expect("tempdir");
982        let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
983        let id = make_id("2401.12345");
984        let src = make_src(&id);
985        // Write a stale-schema entry manually.
986        let bad = serde_json::json!({
987            "schema_version": "0.9",
988            "fetched_at": Utc::now().to_rfc3339(),
989            "ttl_seconds": 86_400 * 7i64,
990            "inner": src,
991        });
992        let path = cache_file(&root, &id);
993        std::fs::create_dir_all(path.parent().expect("parent")).expect("mkdir");
994        std::fs::write(&path, serde_json::to_vec(&bad).expect("json")).expect("write");
995        assert!(
996            cache_read_at(&root, &id, Utc::now()).is_none(),
997            "stale schema version must be rejected"
998        );
999    }
1000
1001    // ── sanitize_entry_path: zip-slip / traversal guard (ADR-0034 D3) ─────────
1002
1003    #[test]
1004    fn sanitize_accepts_normal_relative_paths() {
1005        assert_eq!(
1006            sanitize_entry_path("main.tex").map(|p| p.as_str().replace('\\', "/")),
1007            Some("main.tex".to_string())
1008        );
1009        assert_eq!(
1010            sanitize_entry_path("figs/diagram.png").map(|p| p.as_str().replace('\\', "/")),
1011            Some("figs/diagram.png".to_string())
1012        );
1013        // `.` segments dropped, `//` collapsed.
1014        assert_eq!(
1015            sanitize_entry_path("./a//b.tex").map(|p| p.as_str().replace('\\', "/")),
1016            Some("a/b.tex".to_string())
1017        );
1018    }
1019
1020    #[test]
1021    fn sanitize_rejects_parent_traversal() {
1022        assert_eq!(sanitize_entry_path("../evil.tex"), None);
1023        assert_eq!(sanitize_entry_path("a/../../etc/passwd"), None);
1024        assert_eq!(sanitize_entry_path("sub/../x"), None);
1025    }
1026
1027    #[test]
1028    fn sanitize_rejects_absolute_and_anchored() {
1029        assert_eq!(sanitize_entry_path("/etc/passwd"), None);
1030        assert_eq!(sanitize_entry_path("\\windows\\system32"), None);
1031        assert_eq!(sanitize_entry_path("C:\\Windows\\evil"), None);
1032        assert_eq!(sanitize_entry_path("C:/Windows/evil"), None);
1033    }
1034
1035    #[test]
1036    fn sanitize_rejects_backslash_traversal_cross_platform() {
1037        // A Windows-style traversal in a Unix-produced tar must be caught
1038        // regardless of the extracting platform.
1039        assert_eq!(sanitize_entry_path("..\\..\\evil"), None);
1040        assert_eq!(sanitize_entry_path("a\\..\\..\\b"), None);
1041    }
1042
1043    #[test]
1044    fn sanitize_rejects_empty_nul_dot_and_colon() {
1045        assert_eq!(sanitize_entry_path(""), None);
1046        assert_eq!(sanitize_entry_path("a/\0/b"), None);
1047        assert_eq!(sanitize_entry_path("."), None); // no normal component
1048        assert_eq!(sanitize_entry_path("a:b/c"), None); // colon in a segment
1049                                                        // ADR-0034 A2 — additional real-world vectors.
1050        assert_eq!(sanitize_entry_path("foo/../../bar"), None); // mid-path escape
1051        assert_eq!(sanitize_entry_path("./.."), None); // leading dot then traversal
1052        assert_eq!(sanitize_entry_path("///"), None); // only separators
1053        assert_eq!(sanitize_entry_path("\\\\"), None); // only backslashes
1054        assert_eq!(sanitize_entry_path("C:evil"), None); // bare drive prefix
1055    }
1056
1057    // ── is_figure ─────────────────────────────────────────────────────────────
1058
1059    #[test]
1060    fn is_figure_matches_allowlist_case_insensitively() {
1061        for f in ["fig.png", "a/b.EPS", "plot.Pdf", "x.svg", "y.JPEG"] {
1062            assert!(is_figure(Utf8Path::new(f)), "{f} should be a figure");
1063        }
1064        for nf in ["main.tex", "refs.bib", "macros.sty", "README"] {
1065            assert!(!is_figure(Utf8Path::new(nf)), "{nf} should NOT be a figure");
1066        }
1067    }
1068
1069    // ── extract_bundle ─────────────────────────────────────────────────────────
1070
1071    #[test]
1072    fn extract_bundle_all_returns_every_regular_file() {
1073        let id = make_id("2401.12345");
1074        let payload = tar_gzip(&[
1075            ("paper.tex", b"\\documentclass{article}"),
1076            ("refs.bib", b"@article{x,title={t}}"),
1077            ("figs/plot.png", b"\x89PNG\r\n"),
1078        ]);
1079        let files = extract_bundle(&id, &payload, BundleFilter::All).expect("bundle");
1080        let mut names: Vec<String> = files
1081            .iter()
1082            .map(|f| f.path.as_str().replace('\\', "/"))
1083            .collect();
1084        names.sort();
1085        assert_eq!(names, vec!["figs/plot.png", "paper.tex", "refs.bib"]);
1086        // Postcondition: every returned path is relative with no traversal.
1087        assert!(files
1088            .iter()
1089            .all(|f| !f.path.as_str().starts_with('/') && !f.path.as_str().contains("..")));
1090    }
1091
1092    #[test]
1093    fn extract_bundle_figures_only_keeps_images() {
1094        let id = make_id("2401.12345");
1095        let payload = tar_gzip(&[
1096            ("paper.tex", b"\\documentclass{article}"),
1097            ("refs.bib", b"@article{x}"),
1098            ("figs/plot.png", b"\x89PNG"),
1099            ("diagram.eps", b"%!PS"),
1100        ]);
1101        let files = extract_bundle(&id, &payload, BundleFilter::FiguresOnly).expect("figs");
1102        let mut names: Vec<String> = files
1103            .iter()
1104            .map(|f| f.path.as_str().replace('\\', "/"))
1105            .collect();
1106        names.sort();
1107        assert_eq!(names, vec!["diagram.eps", "figs/plot.png"]);
1108    }
1109
1110    #[test]
1111    fn extract_bundle_pdf_only_is_source_unavailable() {
1112        let id = make_id("2401.12345");
1113        let err = extract_bundle(&id, b"%PDF-1.5 x", BundleFilter::All).expect_err("pdf-only");
1114        assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1115    }
1116
1117    #[test]
1118    fn extract_bundle_bare_file_is_source_unavailable() {
1119        // A bare (non-gzip, non-PDF) single file is not a bundle (ADR-0034 I6):
1120        // unlike extract_tex (which passes a bare .tex through), the bundle
1121        // path returns SourceUnavailable.
1122        let id = make_id("2401.12345");
1123        let err = extract_bundle(&id, b"\\documentclass{article}\nhi", BundleFilter::All)
1124            .expect_err("bare file is not a bundle");
1125        assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1126    }
1127
1128    #[test]
1129    fn extract_bundle_figures_only_none_present_is_source_unavailable() {
1130        let id = make_id("2401.12345");
1131        let payload = tar_gzip(&[("paper.tex", b"\\documentclass{article}")]);
1132        let err = extract_bundle(&id, &payload, BundleFilter::FiguresOnly).expect_err("no figures");
1133        assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1134    }
1135
1136    #[test]
1137    fn extract_bundle_drops_traversal_entry_via_sanitizer() {
1138        // ADR-0034 I1: prove sanitize_entry_path is WIRED INTO extract_bundle.
1139        // The `tar` *writer* refuses to create a `..` entry, and colon/
1140        // backslash names parse inconsistently across OS tar writers, so we
1141        // hand-build a raw USTAR archive carrying a genuine `../evil.tex` entry
1142        // beside a benign file, then assert the traversal entry is absent from
1143        // the result. If a refactor dropped the sanitize call, this fails.
1144        // (The `..`/absolute/etc. rejections themselves are unit-tested
1145        // directly on `sanitize_entry_path` above.)
1146        fn ustar_block(name: &str, data: &[u8]) -> Vec<u8> {
1147            let mut h = vec![0u8; 512];
1148            h[..name.len()].copy_from_slice(name.as_bytes());
1149            h[100..108].copy_from_slice(b"0000644\0");
1150            h[108..116].copy_from_slice(b"0000000\0");
1151            h[116..124].copy_from_slice(b"0000000\0");
1152            h[124..136].copy_from_slice(format!("{:011o}\0", data.len()).as_bytes());
1153            h[136..148].copy_from_slice(b"00000000000\0");
1154            h[148..156].copy_from_slice(b"        "); // checksum field = 8 spaces
1155            h[156] = b'0'; // typeflag: regular file
1156            h[257..263].copy_from_slice(b"ustar\0");
1157            h[263..265].copy_from_slice(b"00");
1158            let sum: u32 = h.iter().map(|&b| u32::from(b)).sum();
1159            h[148..156].copy_from_slice(format!("{sum:06o}\0 ").as_bytes());
1160            h.extend_from_slice(data);
1161            let pad = (512 - data.len() % 512) % 512;
1162            h.resize(h.len() + pad, 0u8);
1163            h
1164        }
1165        let id = make_id("2401.12345");
1166        let mut tar = ustar_block("../evil.tex", b"evil");
1167        tar.extend(ustar_block("safe.tex", b"\\documentclass{article}"));
1168        tar.resize(tar.len() + 1024, 0u8); // two zero end-of-archive blocks
1169        let gz = gzip_bytes(&tar);
1170
1171        let files = extract_bundle(&id, &gz, BundleFilter::All).expect("bundle");
1172        let names: Vec<String> = files
1173            .iter()
1174            .map(|f| f.path.as_str().replace('\\', "/"))
1175            .collect();
1176        assert!(
1177            names.iter().all(|n| !n.contains("..")),
1178            "traversal entry must be rejected; got {names:?}"
1179        );
1180        assert!(
1181            names.iter().any(|n| n == "safe.tex"),
1182            "benign sibling must survive; got {names:?}"
1183        );
1184    }
1185}