1use camino::{Utf8Path, Utf8PathBuf};
49use chrono::{DateTime, Duration, Utc};
50use flate2::read::GzDecoder;
51use serde::{Deserialize, Serialize};
52use std::io::Read;
53use tar::Archive;
54use url::Url;
55
56use crate::provenance::{Capability, LogEvent, LogResult, RowInput};
57use crate::source::{FetchContext, FetchError};
58use crate::{ArxivId, Ref};
59
60const HTTP_SOURCE_KEY: &str = "arxiv";
62
63const PROV_SOURCE_LABEL: &str = "arxiv-src";
65
66const PROV_SOURCE_BUNDLE_LABEL: &str = "arxiv-src-bundle";
70
71pub const ARXIV_SRC_DEFAULT_BASE: &str = "https://export.arxiv.org";
74
75const TEX_SRC_CACHE_TTL_DAYS: i64 = 7;
78const TEX_SRC_CACHE_SCHEMA_VERSION: &str = "1.0";
79
80#[derive(Debug, Serialize, Deserialize)]
81struct CacheEntry {
82 schema_version: String,
83 fetched_at: String,
86 ttl_seconds: i64,
89 inner: PaperTexSource,
90}
91
92#[derive(Debug)]
98pub(crate) struct ExtractedTex {
99 pub main_file: Option<String>,
102 pub content: String,
104}
105
106#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
108pub struct PaperTexSource {
109 pub arxiv_id: String,
111 pub main_file: Option<String>,
114 pub tex_source: String,
116 pub char_count: usize,
118 pub truncated: bool,
120 pub retrieved_from: String,
122}
123
124pub async fn paper_tex_source(
139 base: &Url,
140 id: &ArxivId,
141 max_chars: Option<usize>,
142 ctx: &FetchContext,
143) -> Result<PaperTexSource, FetchError> {
144 if let Some(root) = &ctx.cache_root {
145 if let Some(full) = cache_read(root, id) {
146 return Ok(apply_max_chars(full, max_chars));
147 }
148 }
149
150 let full = fetch_and_extract(base, id, ctx).await?;
151
152 if let Some(root) = &ctx.cache_root {
153 if !cache_write(root, id, &full) {
154 tracing::warn!(
155 cache_root = %root,
156 arxiv_id = %id.as_str(),
157 "tex-source cache write failed; next request will re-fetch"
158 );
159 }
160 }
161
162 Ok(apply_max_chars(full, max_chars))
163}
164
165async fn fetch_and_extract(
166 base: &Url,
167 id: &ArxivId,
168 ctx: &FetchContext,
169) -> Result<PaperTexSource, FetchError> {
170 let _permit = ctx.rate_limiter.acquire(HTTP_SOURCE_KEY).await;
171
172 let url = src_url(base, id)?;
173 let (body, final_url) = ctx.http.fetch_bytes(HTTP_SOURCE_KEY, url).await?;
174
175 let extracted = extract_tex(id, &body)?;
176 let char_count = extracted.content.chars().count();
177
178 let canonical = Ref::Arxiv(id.clone())
179 .promote(PROV_SOURCE_LABEL, None)
180 .digest_hex();
181 ctx.log.append(RowInput {
182 event: LogEvent::Fetch,
183 result: LogResult::Ok,
184 capability: Capability::Oa,
185 ref_: Some(id.as_str()),
186 source: Some(PROV_SOURCE_LABEL),
187 error_code: None,
188 size_bytes: Some(body.len() as u64),
189 license: Some("arxiv-default"),
190 store_path: None,
191 canonical_digest: Some(&canonical),
192 })?;
193
194 Ok(PaperTexSource {
195 arxiv_id: id.as_str().to_string(),
196 main_file: extracted.main_file,
197 tex_source: extracted.content,
198 char_count,
199 truncated: false,
200 retrieved_from: final_url.to_string(),
201 })
202}
203
204fn src_url(base: &Url, id: &ArxivId) -> Result<Url, FetchError> {
205 base.join(&format!("/src/{}", id.as_str()))
206 .map_err(|e| FetchError::SourceSchema {
207 hint: format!("arXiv src URL construction failed: {e}"),
208 })
209}
210
211#[derive(Debug)]
216enum SrcPayload {
217 PdfOnly,
219 SingleFile(Vec<u8>),
222 Tar(Vec<u8>),
224}
225
226fn classify_src(bytes: &[u8], max_decompressed: Option<u64>) -> Result<SrcPayload, FetchError> {
236 if bytes.starts_with(b"%PDF-") {
237 return Ok(SrcPayload::PdfOnly);
238 }
239 if bytes.len() < 2 || bytes[0..2] != [0x1f, 0x8b] {
241 return Ok(SrcPayload::SingleFile(bytes.to_vec()));
242 }
243 let mut decompressed = Vec::new();
244 match max_decompressed {
245 Some(cap) => {
246 let mut gz = GzDecoder::new(std::io::Cursor::new(bytes)).take(cap + 1);
249 gz.read_to_end(&mut decompressed)
250 .map_err(|e| FetchError::SourceSchema {
251 hint: format!("gzip decompress of arXiv src failed: {e}"),
252 })?;
253 if decompressed.len() as u64 > cap {
254 return Err(FetchError::SourceSchema {
255 hint: format!(
256 "arXiv src decompressed size exceeds {cap} bytes \
257 (possible gzip bomb); refusing"
258 ),
259 });
260 }
261 }
262 None => {
263 let mut gz = GzDecoder::new(std::io::Cursor::new(bytes));
264 gz.read_to_end(&mut decompressed)
265 .map_err(|e| FetchError::SourceSchema {
266 hint: format!("gzip decompress of arXiv src failed: {e}"),
267 })?;
268 }
269 }
270 let is_tar = decompressed.len() > 262 && &decompressed[257..262] == b"ustar";
274 if is_tar {
275 Ok(SrcPayload::Tar(decompressed))
276 } else {
277 Ok(SrcPayload::SingleFile(decompressed))
278 }
279}
280
281pub(crate) fn extract_tex(id: &ArxivId, bytes: &[u8]) -> Result<ExtractedTex, FetchError> {
285 match classify_src(bytes, Some(SRC_MAX_DECOMPRESSED_BYTES))? {
294 SrcPayload::PdfOnly => Err(FetchError::TextUnavailable {
295 arxiv_id: id.clone(),
296 }),
297 SrcPayload::SingleFile(data) => {
298 let text = String::from_utf8_lossy(&data).into_owned();
299 if text.trim().is_empty() {
300 return Err(FetchError::TextUnavailable {
301 arxiv_id: id.clone(),
302 });
303 }
304 Ok(ExtractedTex {
305 main_file: None,
306 content: text,
307 })
308 }
309 SrcPayload::Tar(decompressed) => extract_from_tar(id, &decompressed),
310 }
311}
312
313fn extract_from_tar(id: &ArxivId, bytes: &[u8]) -> Result<ExtractedTex, FetchError> {
328 let mut archive = Archive::new(std::io::Cursor::new(bytes));
329 let entries = archive.entries().map_err(|e| FetchError::SourceSchema {
330 hint: format!("tar read failed: {e}"),
331 })?;
332
333 let mut tex_files: Vec<(String, String)> = Vec::new();
334 let mut tex_attempted: usize = 0;
337 let mut unreadable: usize = 0;
341 for entry in entries {
342 let Ok(mut entry) = entry else {
343 unreadable += 1;
344 continue;
345 };
346 let raw = match entry.path() {
347 Ok(p) => p.to_string_lossy().to_string(),
348 Err(_) => {
349 unreadable += 1;
350 continue;
351 }
352 };
353 let Some(path) = sanitize_entry_path(&raw).map(|p| p.to_string()) else {
358 tracing::warn!(arxiv_id = %id.as_str(), entry = %raw, "skipping unsafe arXiv src entry path");
359 continue;
360 };
361 if !path.ends_with(".tex") {
362 continue;
363 }
364 tex_attempted += 1;
365 let mut content = String::new();
366 match entry.read_to_string(&mut content) {
367 Ok(_) if !content.trim().is_empty() => tex_files.push((path, content)),
368 Ok(_) => {} Err(_) => unreadable += 1,
370 }
371 }
372 if unreadable > 0 {
373 tracing::warn!(
374 arxiv_id = %id.as_str(),
375 unreadable,
376 "some arXiv src tar entries were unreadable/unsafe and were skipped"
377 );
378 }
379
380 if tex_files.is_empty() {
381 return Err(if tex_attempted > 0 {
386 FetchError::SourceSchema {
387 hint: format!("tar contained {tex_attempted} .tex entries but all failed to read"),
388 }
389 } else {
390 FetchError::TextUnavailable {
391 arxiv_id: id.clone(),
392 }
393 });
394 }
395
396 let best = tex_files.into_iter().max_by_key(|(name, content)| {
397 let docclass = i64::from(content.contains(r"\documentclass")) * 1_000_000;
398 let is_main = i64::from(name.ends_with("main.tex") || name == "main.tex") * 100_000;
399 let size = i64::try_from(content.len()).unwrap_or(i64::MAX);
400 docclass.saturating_add(is_main).saturating_add(size)
401 });
402
403 match best {
404 Some((name, content)) => Ok(ExtractedTex {
405 main_file: Some(name),
406 content,
407 }),
408 None => Err(FetchError::TextUnavailable {
409 arxiv_id: id.clone(),
410 }),
411 }
412}
413
414fn apply_max_chars(mut full: PaperTexSource, max_chars: Option<usize>) -> PaperTexSource {
415 let Some(max) = max_chars else {
416 return full;
417 };
418 if full.char_count <= max {
419 return full;
420 }
421 full.tex_source = full.tex_source.chars().take(max).collect();
422 full.char_count = max;
423 full.truncated = true;
424 full
425}
426
427fn cache_file(cache_root: &Utf8Path, id: &ArxivId) -> Utf8PathBuf {
428 let safekey = Ref::Arxiv(id.clone()).safekey();
429 cache_root
430 .join("tex-src")
431 .join(format!("{}.json", safekey.as_str()))
432}
433
434fn cache_read(cache_root: &Utf8Path, id: &ArxivId) -> Option<PaperTexSource> {
435 cache_read_at(cache_root, id, Utc::now())
436}
437
438fn cache_read_at(
439 cache_root: &Utf8Path,
440 id: &ArxivId,
441 now: DateTime<Utc>,
442) -> Option<PaperTexSource> {
443 let path = cache_file(cache_root, id);
444 let bytes = std::fs::read(&path).ok()?;
445 let entry: CacheEntry = serde_json::from_slice(&bytes).ok()?;
446 if entry.schema_version != TEX_SRC_CACHE_SCHEMA_VERSION {
447 return None;
448 }
449 let fetched = DateTime::parse_from_rfc3339(&entry.fetched_at)
450 .ok()?
451 .with_timezone(&Utc);
452 if now.signed_duration_since(fetched) > Duration::seconds(entry.ttl_seconds) {
453 return None;
454 }
455 Some(entry.inner)
456}
457
458fn cache_write(cache_root: &Utf8Path, id: &ArxivId, full: &PaperTexSource) -> bool {
459 cache_write_at(cache_root, id, full, Utc::now())
460}
461
462fn cache_write_at(
463 cache_root: &Utf8Path,
464 id: &ArxivId,
465 full: &PaperTexSource,
466 now: DateTime<Utc>,
467) -> bool {
468 let path = cache_file(cache_root, id);
469 if let Some(dir) = path.parent() {
470 if std::fs::create_dir_all(dir).is_err() {
471 return false;
472 }
473 }
474 let entry = CacheEntry {
475 schema_version: TEX_SRC_CACHE_SCHEMA_VERSION.to_string(),
476 fetched_at: now.to_rfc3339(),
477 ttl_seconds: TEX_SRC_CACHE_TTL_DAYS * 86_400,
478 inner: full.clone(),
479 };
480 match serde_json::to_vec(&entry) {
481 Ok(bytes) => std::fs::write(&path, bytes).is_ok(),
482 Err(_) => false,
483 }
484}
485
486pub fn resolve_arxiv_src_base() -> Result<Url, String> {
488 let raw = std::env::var("DOIGET_ARXIV_SRC_BASE")
489 .unwrap_or_else(|_| ARXIV_SRC_DEFAULT_BASE.to_string());
490 Url::parse(&raw).map_err(|e| format!("DOIGET_ARXIV_SRC_BASE is not a valid URL: {e}"))
491}
492
493const FIGURE_EXTS: &[&str] = &["pdf", "eps", "ps", "png", "jpg", "jpeg", "gif", "svg"];
504
505const SRC_MAX_DECOMPRESSED_BYTES: u64 = 500_000_000;
511
512#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
514pub enum BundleFilter {
515 All,
517 FiguresOnly,
519}
520
521#[derive(Debug, Clone, PartialEq, Eq)]
523#[non_exhaustive]
524pub struct SourceFile {
525 pub(crate) path: Utf8PathBuf,
533 pub bytes: Vec<u8>,
535}
536
537impl SourceFile {
538 #[must_use]
540 pub fn path(&self) -> &Utf8Path {
541 &self.path
542 }
543}
544
545fn sanitize_entry_path(raw: &str) -> Option<Utf8PathBuf> {
555 if raw.is_empty() || raw.contains('\0') {
556 return None;
557 }
558 if raw.starts_with('/') || raw.starts_with('\\') {
560 return None;
561 }
562 let b = raw.as_bytes();
563 if b.len() >= 2 && b[0].is_ascii_alphabetic() && b[1] == b':' {
565 return None;
566 }
567 let mut out = Utf8PathBuf::new();
568 let mut any = false;
569 for seg in raw.split(['/', '\\']) {
570 match seg {
571 "" | "." => continue, ".." => return None, s => {
574 if s.contains(':') || s.contains('\0') {
575 return None;
576 }
577 out.push(s);
578 any = true;
579 }
580 }
581 }
582 if any {
583 Some(out)
584 } else {
585 None
586 }
587}
588
589fn is_figure(path: &Utf8Path) -> bool {
591 match path.extension() {
592 Some(ext) => FIGURE_EXTS.contains(&ext.to_ascii_lowercase().as_str()),
593 None => false,
594 }
595}
596
597pub(crate) fn extract_bundle(
614 id: &ArxivId,
615 bytes: &[u8],
616 filter: BundleFilter,
617) -> Result<Vec<SourceFile>, FetchError> {
618 let decompressed = match classify_src(bytes, Some(SRC_MAX_DECOMPRESSED_BYTES))? {
622 SrcPayload::Tar(d) => d,
623 SrcPayload::PdfOnly | SrcPayload::SingleFile(_) => return Err(no_files(id, filter)),
624 };
625
626 let mut archive = Archive::new(std::io::Cursor::new(decompressed));
627 let entries = archive.entries().map_err(|e| FetchError::SourceSchema {
628 hint: format!("tar read failed: {e}"),
629 })?;
630
631 let mut files: Vec<SourceFile> = Vec::new();
632 let mut unreadable: usize = 0;
638 for entry in entries {
639 let mut entry = match entry {
640 Ok(e) => e,
641 Err(e) => {
642 unreadable += 1;
643 tracing::warn!(arxiv_id = %id.as_str(), error = %e, "arXiv src: skipping malformed tar entry");
644 continue;
645 }
646 };
647 if !entry.header().entry_type().is_file() {
650 continue;
651 }
652 let raw_path = match entry.path() {
653 Ok(p) => p.to_string_lossy().into_owned(),
654 Err(e) => {
655 unreadable += 1;
656 tracing::warn!(arxiv_id = %id.as_str(), error = %e, "arXiv src: tar entry has a non-decodable path; skipping");
657 continue;
658 }
659 };
660 let Some(safe) = sanitize_entry_path(&raw_path) else {
661 tracing::warn!(
662 entry = %raw_path,
663 "arXiv src: rejected unsafe tar entry path (zip-slip guard)"
664 );
665 continue;
666 };
667 if filter == BundleFilter::FiguresOnly && !is_figure(&safe) {
668 continue;
669 }
670 let mut buf = Vec::new();
671 match entry.read_to_end(&mut buf) {
672 Ok(_) => files.push(SourceFile {
673 path: safe,
674 bytes: buf,
675 }),
676 Err(e) => {
677 unreadable += 1;
678 tracing::warn!(arxiv_id = %id.as_str(), entry = %safe, error = %e, "arXiv src: failed to read tar entry; skipping");
679 }
680 }
681 }
682
683 if files.is_empty() {
684 return Err(if unreadable > 0 {
686 FetchError::SourceSchema {
687 hint: format!(
688 "arXiv src tar had {unreadable} unreadable entr(y/ies) and no usable files"
689 ),
690 }
691 } else {
692 no_files(id, filter)
693 });
694 }
695 if unreadable > 0 {
696 tracing::warn!(
697 arxiv_id = %id.as_str(),
698 unreadable,
699 extracted = files.len(),
700 "arXiv src: bundle is partial — some entries were unreadable and skipped"
701 );
702 }
703 Ok(files)
704}
705
706fn no_files(id: &ArxivId, filter: BundleFilter) -> FetchError {
710 FetchError::SourceUnavailable {
711 arxiv_id: id.clone(),
712 kind: match filter {
713 BundleFilter::All => "source bundle",
714 BundleFilter::FiguresOnly => "figures",
715 },
716 }
717}
718
719pub async fn paper_source_bundle(
734 base: &Url,
735 id: &ArxivId,
736 filter: BundleFilter,
737 ctx: &FetchContext,
738) -> Result<Vec<SourceFile>, FetchError> {
739 let _permit = ctx.rate_limiter.acquire(HTTP_SOURCE_KEY).await;
740
741 let url = src_url(base, id)?;
742 let (body, _final_url) = ctx.http.fetch_bytes(HTTP_SOURCE_KEY, url).await?;
743
744 let files = extract_bundle(id, &body, filter)?;
745
746 let canonical = Ref::Arxiv(id.clone())
747 .promote(PROV_SOURCE_BUNDLE_LABEL, None)
748 .digest_hex();
749 ctx.log.append(RowInput {
750 event: LogEvent::Fetch,
751 result: LogResult::Ok,
752 capability: Capability::Oa,
753 ref_: Some(id.as_str()),
754 source: Some(PROV_SOURCE_BUNDLE_LABEL),
755 error_code: None,
756 size_bytes: Some(body.len() as u64),
757 license: Some("arxiv-default"),
758 store_path: None,
759 canonical_digest: Some(&canonical),
760 })?;
761
762 Ok(files)
763}
764
765#[cfg(test)]
766#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic, missing_docs)]
767mod tests {
768 use super::*;
769 use flate2::write::GzEncoder;
770 use flate2::Compression;
771 use std::io::Write as _;
772
773 fn make_id(s: &str) -> ArxivId {
774 match Ref::parse(s).expect("parse") {
775 Ref::Arxiv(a) => a,
776 _ => panic!("expected arxiv id"),
777 }
778 }
779
780 fn gzip_bytes(data: &[u8]) -> Vec<u8> {
781 let mut enc = GzEncoder::new(Vec::new(), Compression::default());
782 enc.write_all(data).expect("gzip write");
783 enc.finish().expect("gzip finish")
784 }
785
786 fn tar_gzip(files: &[(&str, &[u8])]) -> Vec<u8> {
787 let mut builder = tar::Builder::new(Vec::new());
788 for (name, data) in files {
789 let mut header = tar::Header::new_gnu();
790 header.set_size(data.len() as u64);
791 header.set_mode(0o644);
792 header.set_cksum();
793 builder
794 .append_data(&mut header, name, std::io::Cursor::new(data))
795 .expect("tar append");
796 }
797 gzip_bytes(&builder.into_inner().expect("tar finish"))
798 }
799
800 fn make_src(id: &ArxivId) -> PaperTexSource {
801 PaperTexSource {
802 arxiv_id: id.as_str().to_string(),
803 main_file: Some("main.tex".into()),
804 tex_source: "\\documentclass{article}".into(),
805 char_count: 23,
806 truncated: false,
807 retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
808 }
809 }
810
811 #[test]
814 fn apply_max_chars_no_cap_is_identity() {
815 let id = make_id("2401.12345");
816 let src = make_src(&id);
817 let out = apply_max_chars(src.clone(), None);
818 assert_eq!(out, src);
819 }
820
821 #[test]
822 fn apply_max_chars_truncates() {
823 let id = make_id("2401.12345");
824 let src = PaperTexSource {
825 arxiv_id: id.as_str().to_string(),
826 main_file: None,
827 tex_source: "abcdefghij".into(),
828 char_count: 10,
829 truncated: false,
830 retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
831 };
832 let out = apply_max_chars(src, Some(4));
833 assert_eq!(out.tex_source, "abcd");
834 assert_eq!(out.char_count, 4);
835 assert!(out.truncated);
836 }
837
838 #[test]
841 fn pdf_only_yields_text_unavailable() {
842 let id = make_id("2401.12345");
843 let result = extract_tex(&id, b"%PDF-1.4 fake");
844 assert!(matches!(result, Err(FetchError::TextUnavailable { .. })));
845 }
846
847 #[test]
848 fn raw_tex_passthrough() {
849 let id = make_id("2401.12345");
850 let tex = b"\\documentclass{article}\n\\begin{document}\nHello.\\end{document}";
851 let ext = extract_tex(&id, tex).expect("extract");
852 assert!(ext.main_file.is_none());
853 assert!(ext.content.contains("\\documentclass"));
854 }
855
856 #[test]
857 fn gzip_single_file_extracted() {
858 let id = make_id("2401.12345");
859 let tex = b"\\documentclass{article}\n\\begin{document}Hello\\end{document}";
860 let gz = gzip_bytes(tex);
861 let ext = extract_tex(&id, &gz).expect("extract");
862 assert!(ext.main_file.is_none(), "single gzip has no tar filename");
863 assert!(ext.content.contains("\\documentclass"));
864 }
865
866 #[test]
869 fn classify_src_rejects_decompression_over_cap() {
870 let big = vec![b'x'; 10_000];
876 let gz = gzip_bytes(&big);
877 let err = classify_src(&gz, Some(1_000)).expect_err("over-cap must be rejected");
878 assert!(
879 matches!(err, FetchError::SourceSchema { .. }),
880 "got {err:?}"
881 );
882 }
883
884 #[test]
885 fn classify_src_accepts_decompression_within_cap() {
886 let small = vec![b'x'; 500];
887 let gz = gzip_bytes(&small);
888 let payload = classify_src(&gz, Some(1_000)).expect("within cap");
889 assert!(matches!(payload, SrcPayload::SingleFile(_)));
890 }
891
892 #[test]
895 fn tar_selects_documentclass_file_over_plain() {
896 let id = make_id("2401.12345");
897 let payload = tar_gzip(&[
898 ("paper.tex", b"\\documentclass{article} main content"),
899 ("macros.tex", b"\\newcommand{\\foo}{bar}"),
900 ]);
901 let ext = extract_tex(&id, &payload).expect("extract");
902 assert_eq!(ext.main_file.as_deref(), Some("paper.tex"));
903 assert!(ext.content.contains("\\documentclass"));
904 }
905
906 #[test]
907 fn tar_prefers_main_tex_among_documentclass_files() {
908 let id = make_id("2401.12345");
909 let payload = tar_gzip(&[
910 ("other.tex", b"\\documentclass{article} other content here"),
911 ("main.tex", b"\\documentclass{article} main"),
912 ]);
913 let ext = extract_tex(&id, &payload).expect("extract");
914 assert_eq!(
915 ext.main_file.as_deref(),
916 Some("main.tex"),
917 "main.tex bonus must override smaller-but-also-documentclass other.tex"
918 );
919 }
920
921 #[test]
922 fn tar_falls_back_to_largest_file_when_no_documentclass() {
923 let id = make_id("2401.12345");
924 let short = b"\\section{Short}".as_slice();
925 let mut long_content = b"\\section{Long} ".to_vec();
926 long_content.extend(vec![b'x'; 500]);
927 let payload = tar_gzip(&[("short.tex", short), ("long.tex", &long_content)]);
928 let ext = extract_tex(&id, &payload).expect("extract");
929 assert_eq!(ext.main_file.as_deref(), Some("long.tex"));
930 }
931
932 #[test]
933 fn tar_with_no_tex_files_is_text_unavailable() {
934 let id = make_id("2401.12345");
935 let payload = tar_gzip(&[("README.md", b"# Paper"), ("figure.eps", b"%!PS")]);
936 let err = extract_tex(&id, &payload).expect_err("should fail");
937 assert!(matches!(err, FetchError::TextUnavailable { .. }));
938 }
939
940 #[test]
943 fn resolve_base_defaults_to_production() {
944 if std::env::var("DOIGET_ARXIV_SRC_BASE").is_err() {
945 let u = resolve_arxiv_src_base().expect("resolve");
946 assert_eq!(u.as_str(), "https://export.arxiv.org/");
947 }
948 }
949
950 #[test]
951 fn cache_round_trip() {
952 let dir = tempfile::tempdir().expect("tempdir");
953 let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
954 let id = make_id("2401.12345");
955 let src = make_src(&id);
956 assert!(cache_write(&root, &id, &src));
957 let read = cache_read(&root, &id).expect("cache hit");
958 assert_eq!(read, src);
959 }
960
961 #[test]
962 fn cache_expired_returns_none() {
963 let dir = tempfile::tempdir().expect("tempdir");
964 let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
965 let id = make_id("2401.12345");
966 let src = PaperTexSource {
967 arxiv_id: id.as_str().to_string(),
968 main_file: None,
969 tex_source: "test".into(),
970 char_count: 4,
971 truncated: false,
972 retrieved_from: "https://export.arxiv.org/src/2401.12345".into(),
973 };
974 let past = Utc::now() - Duration::days(TEX_SRC_CACHE_TTL_DAYS + 1);
975 assert!(cache_write_at(&root, &id, &src, past));
976 assert!(cache_read_at(&root, &id, Utc::now()).is_none());
977 }
978
979 #[test]
980 fn cache_schema_version_mismatch_returns_none() {
981 let dir = tempfile::tempdir().expect("tempdir");
982 let root = camino::Utf8PathBuf::from_path_buf(dir.path().to_path_buf()).expect("utf8");
983 let id = make_id("2401.12345");
984 let src = make_src(&id);
985 let bad = serde_json::json!({
987 "schema_version": "0.9",
988 "fetched_at": Utc::now().to_rfc3339(),
989 "ttl_seconds": 86_400 * 7i64,
990 "inner": src,
991 });
992 let path = cache_file(&root, &id);
993 std::fs::create_dir_all(path.parent().expect("parent")).expect("mkdir");
994 std::fs::write(&path, serde_json::to_vec(&bad).expect("json")).expect("write");
995 assert!(
996 cache_read_at(&root, &id, Utc::now()).is_none(),
997 "stale schema version must be rejected"
998 );
999 }
1000
1001 #[test]
1004 fn sanitize_accepts_normal_relative_paths() {
1005 assert_eq!(
1006 sanitize_entry_path("main.tex").map(|p| p.as_str().replace('\\', "/")),
1007 Some("main.tex".to_string())
1008 );
1009 assert_eq!(
1010 sanitize_entry_path("figs/diagram.png").map(|p| p.as_str().replace('\\', "/")),
1011 Some("figs/diagram.png".to_string())
1012 );
1013 assert_eq!(
1015 sanitize_entry_path("./a//b.tex").map(|p| p.as_str().replace('\\', "/")),
1016 Some("a/b.tex".to_string())
1017 );
1018 }
1019
1020 #[test]
1021 fn sanitize_rejects_parent_traversal() {
1022 assert_eq!(sanitize_entry_path("../evil.tex"), None);
1023 assert_eq!(sanitize_entry_path("a/../../etc/passwd"), None);
1024 assert_eq!(sanitize_entry_path("sub/../x"), None);
1025 }
1026
1027 #[test]
1028 fn sanitize_rejects_absolute_and_anchored() {
1029 assert_eq!(sanitize_entry_path("/etc/passwd"), None);
1030 assert_eq!(sanitize_entry_path("\\windows\\system32"), None);
1031 assert_eq!(sanitize_entry_path("C:\\Windows\\evil"), None);
1032 assert_eq!(sanitize_entry_path("C:/Windows/evil"), None);
1033 }
1034
1035 #[test]
1036 fn sanitize_rejects_backslash_traversal_cross_platform() {
1037 assert_eq!(sanitize_entry_path("..\\..\\evil"), None);
1040 assert_eq!(sanitize_entry_path("a\\..\\..\\b"), None);
1041 }
1042
1043 #[test]
1044 fn sanitize_rejects_empty_nul_dot_and_colon() {
1045 assert_eq!(sanitize_entry_path(""), None);
1046 assert_eq!(sanitize_entry_path("a/\0/b"), None);
1047 assert_eq!(sanitize_entry_path("."), None); assert_eq!(sanitize_entry_path("a:b/c"), None); assert_eq!(sanitize_entry_path("foo/../../bar"), None); assert_eq!(sanitize_entry_path("./.."), None); assert_eq!(sanitize_entry_path("///"), None); assert_eq!(sanitize_entry_path("\\\\"), None); assert_eq!(sanitize_entry_path("C:evil"), None); }
1056
1057 #[test]
1060 fn is_figure_matches_allowlist_case_insensitively() {
1061 for f in ["fig.png", "a/b.EPS", "plot.Pdf", "x.svg", "y.JPEG"] {
1062 assert!(is_figure(Utf8Path::new(f)), "{f} should be a figure");
1063 }
1064 for nf in ["main.tex", "refs.bib", "macros.sty", "README"] {
1065 assert!(!is_figure(Utf8Path::new(nf)), "{nf} should NOT be a figure");
1066 }
1067 }
1068
1069 #[test]
1072 fn extract_bundle_all_returns_every_regular_file() {
1073 let id = make_id("2401.12345");
1074 let payload = tar_gzip(&[
1075 ("paper.tex", b"\\documentclass{article}"),
1076 ("refs.bib", b"@article{x,title={t}}"),
1077 ("figs/plot.png", b"\x89PNG\r\n"),
1078 ]);
1079 let files = extract_bundle(&id, &payload, BundleFilter::All).expect("bundle");
1080 let mut names: Vec<String> = files
1081 .iter()
1082 .map(|f| f.path.as_str().replace('\\', "/"))
1083 .collect();
1084 names.sort();
1085 assert_eq!(names, vec!["figs/plot.png", "paper.tex", "refs.bib"]);
1086 assert!(files
1088 .iter()
1089 .all(|f| !f.path.as_str().starts_with('/') && !f.path.as_str().contains("..")));
1090 }
1091
1092 #[test]
1093 fn extract_bundle_figures_only_keeps_images() {
1094 let id = make_id("2401.12345");
1095 let payload = tar_gzip(&[
1096 ("paper.tex", b"\\documentclass{article}"),
1097 ("refs.bib", b"@article{x}"),
1098 ("figs/plot.png", b"\x89PNG"),
1099 ("diagram.eps", b"%!PS"),
1100 ]);
1101 let files = extract_bundle(&id, &payload, BundleFilter::FiguresOnly).expect("figs");
1102 let mut names: Vec<String> = files
1103 .iter()
1104 .map(|f| f.path.as_str().replace('\\', "/"))
1105 .collect();
1106 names.sort();
1107 assert_eq!(names, vec!["diagram.eps", "figs/plot.png"]);
1108 }
1109
1110 #[test]
1111 fn extract_bundle_pdf_only_is_source_unavailable() {
1112 let id = make_id("2401.12345");
1113 let err = extract_bundle(&id, b"%PDF-1.5 x", BundleFilter::All).expect_err("pdf-only");
1114 assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1115 }
1116
1117 #[test]
1118 fn extract_bundle_bare_file_is_source_unavailable() {
1119 let id = make_id("2401.12345");
1123 let err = extract_bundle(&id, b"\\documentclass{article}\nhi", BundleFilter::All)
1124 .expect_err("bare file is not a bundle");
1125 assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1126 }
1127
1128 #[test]
1129 fn extract_bundle_figures_only_none_present_is_source_unavailable() {
1130 let id = make_id("2401.12345");
1131 let payload = tar_gzip(&[("paper.tex", b"\\documentclass{article}")]);
1132 let err = extract_bundle(&id, &payload, BundleFilter::FiguresOnly).expect_err("no figures");
1133 assert!(matches!(err, FetchError::SourceUnavailable { .. }));
1134 }
1135
1136 #[test]
1137 fn extract_bundle_drops_traversal_entry_via_sanitizer() {
1138 fn ustar_block(name: &str, data: &[u8]) -> Vec<u8> {
1147 let mut h = vec![0u8; 512];
1148 h[..name.len()].copy_from_slice(name.as_bytes());
1149 h[100..108].copy_from_slice(b"0000644\0");
1150 h[108..116].copy_from_slice(b"0000000\0");
1151 h[116..124].copy_from_slice(b"0000000\0");
1152 h[124..136].copy_from_slice(format!("{:011o}\0", data.len()).as_bytes());
1153 h[136..148].copy_from_slice(b"00000000000\0");
1154 h[148..156].copy_from_slice(b" "); h[156] = b'0'; h[257..263].copy_from_slice(b"ustar\0");
1157 h[263..265].copy_from_slice(b"00");
1158 let sum: u32 = h.iter().map(|&b| u32::from(b)).sum();
1159 h[148..156].copy_from_slice(format!("{sum:06o}\0 ").as_bytes());
1160 h.extend_from_slice(data);
1161 let pad = (512 - data.len() % 512) % 512;
1162 h.resize(h.len() + pad, 0u8);
1163 h
1164 }
1165 let id = make_id("2401.12345");
1166 let mut tar = ustar_block("../evil.tex", b"evil");
1167 tar.extend(ustar_block("safe.tex", b"\\documentclass{article}"));
1168 tar.resize(tar.len() + 1024, 0u8); let gz = gzip_bytes(&tar);
1170
1171 let files = extract_bundle(&id, &gz, BundleFilter::All).expect("bundle");
1172 let names: Vec<String> = files
1173 .iter()
1174 .map(|f| f.path.as_str().replace('\\', "/"))
1175 .collect();
1176 assert!(
1177 names.iter().all(|n| !n.contains("..")),
1178 "traversal entry must be rejected; got {names:?}"
1179 );
1180 assert!(
1181 names.iter().any(|n| n == "safe.tex"),
1182 "benign sibling must survive; got {names:?}"
1183 );
1184 }
1185}