Skip to main content

doiget_core/
refs.rs

1//! Bibliography input adapters per ADR-0030.
2//!
3//! Parses three input shapes into an iterator of `Ref`s with optional
4//! `entry_key` provenance back to the source bibliography:
5//!
6//! - **Plain refs**: one `doi:…` / `arxiv:…` / bare-DOI / bare-arXiv id
7//!   per line, with `#`-prefixed comments and blank lines tolerated.
8//!   The existing `doiget batch <refs.txt>` shape.
9//! - **CSL-JSON**: a JSON array of entries with `id` (citation key),
10//!   `DOI`, and optionally `archivePrefix = "arXiv"` + `eprint`
11//!   fields. Parsed via the workspace's existing `serde_json` — no
12//!   new dependency.
13//! - **BibTeX / BibLaTeX (.bib)**: parsed via the `biblatex` crate
14//!   (ADR-0030 D2). One `@entrytype{KEY, …}` per entry; the `doi`
15//!   field is preferred, falling back to an arXiv `eprint`.
16//!
17//! Identifier-pick priority per ADR-0030 D3: `doi` > `arxiv` > `pmid`.
18//! A PMID / PMCID entry is reported here as `UnsupportedIdentifier` and
19//! turned into the DOI PubMed lists for it by
20//! [`crate::pubmed::resolve_entries`] (#500, ADR-0061) -- a request, so
21//! not in this pure parser. There is no `Ref::Pmid`: the DOI is the
22//! identity.
23//!
24//! Parse-error policy per ADR-0030 D5: a single entry's failure is
25//! captured per-entry and does NOT abort the whole batch. The caller
26//! decides whether to skip-and-warn (default) or fail-closed
27//! (`--strict`).
28
29use biblatex::{Bibliography, ChunksExt};
30use camino::Utf8Path;
31use thiserror::Error;
32
33use crate::{Ref, RefParseError};
34
35/// One successfully-parsed bibliography entry.
36///
37/// `entry_key` echoes the source bibliography's citation key
38/// (BibTeX `@article{KEY,…}` / CSL-JSON `"id"`) so downstream
39/// automation can bridge the fetch outcome back to the originating
40/// reference — the load-bearing field for the Zotero / Mendeley
41/// "attach fetched PDF to this reference" workflow per ADR-0030 §6.
42#[derive(Debug, Clone, PartialEq, Eq)]
43#[non_exhaustive]
44pub struct ParsedEntry {
45    /// The identifier the adapter chose for this entry (`Ref::Doi` /
46    /// `Ref::Arxiv`).
47    pub ref_: Ref,
48    /// The source bibliography's citation key, when one is available.
49    /// `None` for plain-refs input (no key concept) and for any
50    /// future input shape that lacks per-entry keys.
51    pub entry_key: Option<String>,
52}
53
54/// Why a single bibliography entry failed to produce a `Ref`.
55///
56/// Closed-enum so the failure-class can be exposed at the
57/// `docs/ERRORS.md` §3 INVALID_REF surface without leaking parser
58/// internals.
59#[derive(Debug, Clone, Error, PartialEq, Eq)]
60#[non_exhaustive]
61pub enum ParseError {
62    /// The line did not contain a `doi:` / `arxiv:` / bare-DOI /
63    /// bare-arXiv id — empty (after trimming) or just a comment.
64    /// Plain-refs path filters these out silently; CSL-JSON path
65    /// emits this when an entry has no resolvable identifier.
66    #[error("entry has no DOI / arXiv id (entry_key={entry_key:?})")]
67    NoIdentifier {
68        /// The source bibliography's citation key, when known.
69        entry_key: Option<String>,
70    },
71    /// The entry DOES carry an identifier, and it is one doiget recognises
72    /// and resolves only with a request: a PMID / PMCID, through the DOI
73    /// PubMed lists for it (#500, `crate::pubmed::resolve_entries`).
74    ///
75    /// Distinct from [`Self::NoIdentifier`] because the two send a reader in
76    /// opposite directions. "entry has no DOI / arXiv id" is accurate about
77    /// what the parser did and wrong about the entry: a PubMed-exported
78    /// `.bib` record carrying `pmid = {9659853}` is not deficient, and a user
79    /// who believes it is will go and edit a bibliography that was fine. The
80    /// missing piece is on doiget's side.
81    ///
82    /// Surfaces as `NOT_IMPLEMENTED` rather than `INVALID_REF`: the input is
83    /// valid and the support is absent, and the two carry different advice --
84    /// "wait for a release" versus "correct your input" (ADR-0055).
85    #[error("entry {entry_key:?} is identified only by {kind} {value:?}, which doiget resolves through the DOI PubMed lists for it -- a request this run did not make; it is NOT missing an identifier")]
86    UnsupportedIdentifier {
87        /// Human-facing name of the identifier class, e.g. `"PMID"`.
88        kind: &'static str,
89        /// The identifier as written in the entry.
90        value: String,
91        /// The source bibliography's citation key, when known.
92        entry_key: Option<String>,
93    },
94    /// The entry has no DOI or arXiv id but is software on GitHub: its
95    /// `url` is a repository or release (#614). `verify` checks it still
96    /// resolves, and `doiget cite <url>` renders it as `@software`.
97    #[error("entry {entry_key:?} is software at {url}, with no DOI / arXiv id; `doiget cite {url}` cites it")]
98    SoftwareUrl {
99        /// The GitHub URL as written in the entry.
100        url: String,
101        /// The source bibliography's citation key, when known.
102        entry_key: Option<String>,
103    },
104    /// The identifier was present but `Ref::parse` rejected it
105    /// (malformed DOI suffix, invalid arXiv id shape, etc.).
106    #[error(
107        "entry identifier {raw:?} did not parse as a Ref \
108         (entry_key={entry_key:?}): {source}"
109    )]
110    InvalidRef {
111        /// The raw identifier string the parser saw.
112        raw: String,
113        /// The source bibliography's citation key, when known.
114        entry_key: Option<String>,
115        /// The structured `Ref::parse` failure.
116        #[source]
117        source: RefParseError,
118    },
119    /// The whole input did not deserialise — CSL-JSON that is not a
120    /// JSON array, top-level malformed JSON, etc. This is a
121    /// whole-input failure, not a per-entry failure; callers receive
122    /// it as the sole `Err` element of the result iterator.
123    #[error("input did not deserialise as {format}: {message}")]
124    Decode {
125        /// Which parser branch produced the failure (`"csl-json"` /
126        /// `"bibtex"`).
127        format: &'static str,
128        /// `serde_json::Error::to_string()`.
129        message: String,
130    },
131    /// Format requested or detected, but no parser for it is shipped.
132    /// Retained as a forward-compatible variant for input shapes not
133    /// yet implemented (e.g. a future RIS adapter); the `bibtex` and
134    /// `csl-json` paths are both live today.
135    #[error("{format} parsing is not yet implemented")]
136    UnsupportedFormat {
137        /// The format token naming the unsupported shape.
138        format: &'static str,
139    },
140}
141
142/// The claim [`ParseError::UnsupportedIdentifier`] makes, without the
143/// `entry {entry_key:?}` prefix its `Display` carries.
144///
145/// Callers that put `entry_key` in a field of its own -- the CLI `verify`
146/// row and the MCP `batch_from_bibliography` envelope both do -- would
147/// otherwise say it twice. One definition rather than a copy at each site,
148/// because the copies drifted: two of the three carried a run of joined-line
149/// whitespace into user-facing output before anything asserted the text.
150#[must_use]
151pub fn unsupported_identifier_claim(kind: &str, value: &str) -> String {
152    format!("entry is identified only by {kind} {value:?}, which doiget resolves through the DOI PubMed lists for it (#500) -- a request this run did not make (--dry-run / --offline); it is NOT missing an identifier")
153}
154
155/// Input-shape discriminator per ADR-0030 D4.
156///
157/// `Auto` means "detect from path extension and/or content
158/// fingerprint"; the explicit variants name a parser directly and
159/// skip detection.
160#[derive(Debug, Clone, Copy, PartialEq, Eq)]
161#[non_exhaustive]
162pub enum Format {
163    /// Detect from file extension if a path was supplied, else from
164    /// content fingerprint; fall through to [`Format::Refs`].
165    Auto,
166    /// Plain refs — one identifier per line, `#` comments, blanks.
167    Refs,
168    /// CSL-JSON array per <https://citationstyles.org/>.
169    CslJson,
170    /// BibTeX / BibLaTeX, parsed via the `biblatex` crate (ADR-0030 D2).
171    Bibtex,
172}
173
174impl Format {
175    /// Wire token used by the CLI `--format` flag and the MCP tool
176    /// input schema's `format` field per ADR-0030 §6.
177    pub fn as_wire(&self) -> &'static str {
178        match self {
179            Format::Auto => "auto",
180            Format::Refs => "refs",
181            Format::CslJson => "csl-json",
182            Format::Bibtex => "bibtex",
183        }
184    }
185}
186
187/// Detect the input format per ADR-0030 D4.
188///
189/// Precedence: file extension first (when `path` is `Some`), then
190/// content fingerprint, then fallback to [`Format::Refs`]. The
191/// caller's explicit `--format` flag should short-circuit this
192/// function — it is the slowest of the three precedence rules in the
193/// ADR.
194pub fn detect_format(path: Option<&Utf8Path>, content: &str) -> Format {
195    if let Some(p) = path {
196        let ext = p.extension().unwrap_or_default().to_ascii_lowercase();
197        match ext.as_str() {
198            "bib" | "biblatex" => return Format::Bibtex,
199            "json" | "csl" => return Format::CslJson,
200            _ => {}
201        }
202    }
203    // Content fingerprint: peek the first non-blank, non-comment line.
204    for line in content.lines() {
205        let trimmed = line.trim();
206        if trimmed.is_empty() || trimmed.starts_with('#') {
207            continue;
208        }
209        if trimmed.starts_with('@') {
210            return Format::Bibtex;
211        }
212        if trimmed.starts_with('[') || trimmed.starts_with('{') {
213            return Format::CslJson;
214        }
215        break;
216    }
217    Format::Refs
218}
219
220/// Parse `text` per `format`, dispatching to the matching shape
221/// parser. `path` is consulted only when `format == Format::Auto` to
222/// drive [`detect_format`].
223///
224/// Returns one element per discovered entry — `Ok` for entries that
225/// produced a `Ref`, `Err` for per-entry failures the caller should
226/// surface as a JSONL `INVALID_REF` line. A whole-input decode
227/// failure ([`ParseError::Decode`]) is returned as a single-element
228/// `Err` so the caller's exit-code path treats it as one parse error
229/// rather than zero.
230pub fn parse_input(
231    text: &str,
232    format: Format,
233    path: Option<&Utf8Path>,
234) -> Vec<Result<ParsedEntry, ParseError>> {
235    let resolved = match format {
236        Format::Auto => detect_format(path, text),
237        other => other,
238    };
239    match resolved {
240        Format::Refs | Format::Auto => parse_plain_refs(text),
241        Format::CslJson => parse_csl_json(text),
242        Format::Bibtex => parse_bibtex(text),
243    }
244}
245
246/// Parse plain refs — the existing batch input format. One ref per
247/// non-blank, non-comment line. `entry_key` is always `None` for this
248/// shape; plain refs have no citation-key concept.
249pub fn parse_plain_refs(text: &str) -> Vec<Result<ParsedEntry, ParseError>> {
250    let mut out = Vec::new();
251    for raw_line in text.lines() {
252        let line = raw_line.trim();
253        if line.is_empty() || line.starts_with('#') {
254            continue;
255        }
256        out.push(match Ref::parse(line) {
257            Ok(ref_) => Ok(ParsedEntry {
258                ref_,
259                entry_key: None,
260            }),
261            Err(e) => Err(ParseError::InvalidRef {
262                raw: line.to_string(),
263                entry_key: None,
264                source: e,
265            }),
266        });
267    }
268    out
269}
270
271/// Parse a CSL-JSON document — a JSON array of objects, each with at
272/// least an `id` (citation key) and one of `DOI`, or `archivePrefix`
273/// + `eprint` (arXiv).
274///
275/// Identifier-pick priority per ADR-0030 D3:
276///
277/// 1. `DOI` field (case-sensitive per the CSL-JSON spec but Zotero
278///    sometimes emits `doi` lowercase — we accept both).
279/// 2. `archivePrefix == "arXiv"` (case-insensitive) + `eprint`
280///    (or `note: "arXiv:..."` shape Zotero emits).
281/// 3. A PMID / PMCID (`PMID`, `PMCID`, or Zotero's `note`) is reported as
282///    `UnsupportedIdentifier`, for [`crate::pubmed::resolve_entries`] to
283///    turn into its DOI (#500, ADR-0061).
284///
285/// `entry_key` is the `id` field verbatim.
286pub fn parse_csl_json(text: &str) -> Vec<Result<ParsedEntry, ParseError>> {
287    let parsed: serde_json::Result<Vec<serde_json::Value>> = serde_json::from_str(text);
288    let entries = match parsed {
289        Ok(arr) => arr,
290        Err(e) => {
291            return vec![Err(ParseError::Decode {
292                format: "csl-json",
293                message: e.to_string(),
294            })]
295        }
296    };
297    let mut out = Vec::with_capacity(entries.len());
298    for entry in entries {
299        // `id` is usually a string in real-world Zotero exports but
300        // the spec allows numeric ids too — stringify either form so
301        // the operator can find the entry in their library.
302        let entry_key = entry.get("id").and_then(|v| {
303            if let Some(s) = v.as_str() {
304                Some(s.to_string())
305            } else if v.is_number() {
306                Some(v.to_string())
307            } else {
308                None
309            }
310        });
311        out.push(parse_csl_entry(&entry, entry_key));
312    }
313    out
314}
315
316/// Pick the highest-priority identifier on a single CSL-JSON entry
317/// and parse it. Honors ADR-0030 D3 priority.
318fn parse_csl_entry(
319    entry: &serde_json::Value,
320    entry_key: Option<String>,
321) -> Result<ParsedEntry, ParseError> {
322    // Priority 1: DOI (both `DOI` per spec and `doi` lowercase per
323    // real-world exports). Zotero emits uppercase; Mendeley sometimes
324    // lowercase.
325    if let Some(doi) = entry
326        .get("DOI")
327        .or_else(|| entry.get("doi"))
328        .and_then(|v| v.as_str())
329    {
330        let raw = doi.trim();
331        if !raw.is_empty() {
332            return match Ref::parse(raw) {
333                Ok(ref_) => Ok(ParsedEntry { ref_, entry_key }),
334                Err(e) => Err(ParseError::InvalidRef {
335                    raw: raw.to_string(),
336                    entry_key,
337                    source: e,
338                }),
339            };
340        }
341    }
342    // Priority 2: arXiv — `archivePrefix == "arXiv"` (CSL extension)
343    // OR the Zotero-specific `note: "arXiv:..."` shape.
344    let is_arxiv = entry
345        .get("archivePrefix")
346        .or_else(|| entry.get("archive_prefix"))
347        .and_then(|v| v.as_str())
348        .map(|s| s.eq_ignore_ascii_case("arxiv"))
349        .unwrap_or(false);
350    if is_arxiv {
351        if let Some(eprint) = entry.get("eprint").and_then(|v| v.as_str()) {
352            let raw = eprint.trim();
353            if !raw.is_empty() {
354                let with_scheme = if raw.to_ascii_lowercase().starts_with("arxiv:") {
355                    raw.to_string()
356                } else {
357                    format!("arxiv:{raw}")
358                };
359                return match Ref::parse(&with_scheme) {
360                    Ok(ref_) => Ok(ParsedEntry { ref_, entry_key }),
361                    Err(e) => Err(ParseError::InvalidRef {
362                        raw: with_scheme,
363                        entry_key,
364                        source: e,
365                    }),
366                };
367            }
368        }
369    }
370    // Fallback: scan `note` for an embedded `arXiv:NNNN.NNNNN` —
371    // Zotero often stores the arXiv id there instead of a typed
372    // field. The pattern is intentionally narrow (must follow the
373    // canonical "arXiv:" prefix); free-text DOIs in notes are NOT
374    // mined here.
375    if let Some(note) = entry.get("note").and_then(|v| v.as_str()) {
376        if let Some(idx) = note.to_ascii_lowercase().find("arxiv:") {
377            let tail = &note[idx + "arxiv:".len()..];
378            // Take chars matching the arXiv id alphabet (digits / dot /
379            // slash / letters / hyphen) — stop at the first separator
380            // so the rest of the note is ignored.
381            let id: String = tail
382                .chars()
383                .take_while(|c| matches!(c, '0'..='9' | '.' | '/' | 'a'..='z' | 'A'..='Z' | '-'))
384                .collect();
385            if !id.is_empty() {
386                let with_scheme = format!("arxiv:{id}");
387                return match Ref::parse(&with_scheme) {
388                    Ok(ref_) => Ok(ParsedEntry { ref_, entry_key }),
389                    Err(e) => Err(ParseError::InvalidRef {
390                        raw: with_scheme,
391                        entry_key,
392                        source: e,
393                    }),
394                };
395            }
396        }
397    }
398    // #500, CSL-JSON half. The BibTeX parser learned to say "this entry HAS an
399    // identifier I cannot use" and this one did not, so the same PubMed record
400    // exported as CSL-JSON still got "entry has no DOI / arXiv id" -- the claim
401    // #500 exists to stop doiget making, surviving in the format a Zotero user
402    // is most likely to hand it.
403    if let Some((kind, value)) = csl_unsupported_identifier(entry) {
404        return Err(ParseError::UnsupportedIdentifier {
405            kind,
406            value,
407            entry_key,
408        });
409    }
410    if let Some(url) = entry
411        .get("URL")
412        .and_then(|v| v.as_str())
413        .filter(|u| crate::software::GithubRef::parse(u).is_some())
414    {
415        return Err(ParseError::SoftwareUrl {
416            url: url.trim().to_string(),
417            entry_key,
418        });
419    }
420    Err(ParseError::NoIdentifier { entry_key })
421}
422
423/// The CSL-JSON counterpart of [`unsupported_identifier`]: an identifier
424/// doiget recognises and cannot resolve yet (#500).
425///
426/// CSL-JSON has no standard PMID field, so exporters improvise. Zotero writes
427/// `PMID: 9659853` into `note`; some tools emit a top-level `PMID` key. Both
428/// are checked, and the note scan mirrors the arXiv one directly above it.
429fn csl_unsupported_identifier(entry: &serde_json::Value) -> Option<(&'static str, String)> {
430    let direct = |name: &str| -> Option<String> {
431        let v = entry.get(name)?;
432        let s = match v {
433            serde_json::Value::String(s) => s.trim().to_string(),
434            serde_json::Value::Number(n) => n.to_string(),
435            _ => return None,
436        };
437        (!s.is_empty()).then_some(s)
438    };
439    for (field, kind) in [
440        ("PMID", "PMID"),
441        ("pmid", "PMID"),
442        ("PMCID", "PMCID"),
443        ("pmcid", "PMCID"),
444    ] {
445        if let Some(v) = direct(field) {
446            return Some((kind, v));
447        }
448    }
449
450    // Zotero's `note` carries `PMID: 9659853` / `PMCID: PMC1234567`.
451    let note = entry.get("note").and_then(|v| v.as_str())?;
452    for (needle, kind) in [("pmcid:", "PMCID"), ("pmid:", "PMID")] {
453        let lower = note.to_ascii_lowercase();
454        if let Some(at) = lower.find(needle) {
455            let value: String = note[at + needle.len()..]
456                .trim_start()
457                .chars()
458                .take_while(|c| c.is_ascii_alphanumeric())
459                .collect();
460            if !value.is_empty() {
461                return Some((kind, value));
462            }
463        }
464    }
465    None
466}
467
468/// Parse a BibTeX / BibLaTeX document via the `biblatex` crate
469/// (ADR-0030 D2). One `@entrytype{KEY, …}` produces one entry;
470/// `entry_key` is the citation key verbatim.
471///
472/// A whole-input parse failure (malformed BibTeX the `biblatex` crate
473/// rejects) is returned as a single-element [`ParseError::Decode`] so
474/// the caller's exit-code path counts it as one error rather than
475/// zero — matching the CSL-JSON behaviour.
476///
477/// Identifier-pick priority per ADR-0030 D3: `doi` field, then
478/// `eprint` (arXiv). See `parse_bibtex_entry`.
479pub fn parse_bibtex(text: &str) -> Vec<Result<ParsedEntry, ParseError>> {
480    let bib = match Bibliography::parse(text) {
481        Ok(b) => b,
482        Err(e) => {
483            return vec![Err(ParseError::Decode {
484                format: "bibtex",
485                message: e.to_string(),
486            })]
487        }
488    };
489    bib.iter()
490        .map(|entry| parse_bibtex_entry(entry, Some(entry.key.clone())))
491        .collect()
492}
493
494/// Pick the highest-priority identifier on a single BibTeX entry and
495/// parse it. Honors ADR-0030 D3 priority (`doi` > arXiv `eprint`).
496fn parse_bibtex_entry(
497    entry: &biblatex::Entry,
498    entry_key: Option<String>,
499) -> Result<ParsedEntry, ParseError> {
500    // Priority 1: `doi` field. The typed accessor formats the chunk
501    // value to a `String`; `Err` means the field is absent or not a
502    // plain string, which we treat as "no DOI here" and fall through.
503    if let Ok(doi) = entry.doi() {
504        let raw = doi.trim();
505        if !raw.is_empty() {
506            return match Ref::parse(raw) {
507                Ok(ref_) => Ok(ParsedEntry { ref_, entry_key }),
508                Err(e) => Err(ParseError::InvalidRef {
509                    raw: raw.to_string(),
510                    entry_key,
511                    source: e,
512                }),
513            };
514        }
515    }
516    // Priority 2: arXiv via the `eprint` field. The BibTeX convention
517    // is `eprint = {2204.12345}` with `archivePrefix = {arXiv}` (or the
518    // BibLaTeX `eprinttype = {arxiv}`). We accept the eprint as an arXiv
519    // id when the prefix names arXiv OR is absent (the dominant
520    // single-preprint-server convention); a prefix that names something
521    // else (e.g. `eprinttype = {pubmed}`) is skipped rather than
522    // parsed incorrectly.
523    if let Ok(eprint) = entry.eprint() {
524        let raw = eprint.trim();
525        if !raw.is_empty() && arxiv_eligible(entry) {
526            let with_scheme = if raw.to_ascii_lowercase().starts_with("arxiv:") {
527                raw.to_string()
528            } else {
529                format!("arxiv:{raw}")
530            };
531            return match Ref::parse(&with_scheme) {
532                Ok(ref_) => Ok(ParsedEntry { ref_, entry_key }),
533                Err(e) => Err(ParseError::InvalidRef {
534                    raw: with_scheme,
535                    entry_key,
536                    source: e,
537                }),
538            };
539        }
540    }
541    // #500: before reporting "no identifier", check for one doiget simply
542    // does not support. Saying "no DOI / arXiv id" about an entry that
543    // carries a PMID is accurate about the parser and wrong about the entry.
544    if let Some((kind, value)) = unsupported_identifier(entry) {
545        return Err(ParseError::UnsupportedIdentifier {
546            kind,
547            value,
548            entry_key,
549        });
550    }
551    if let Some(url) = entry
552        .get("url")
553        .map(|v| v.format_verbatim().trim().to_string())
554        .filter(|u| crate::software::GithubRef::parse(u).is_some())
555    {
556        return Err(ParseError::SoftwareUrl { url, entry_key });
557    }
558    Err(ParseError::NoIdentifier { entry_key })
559}
560
561/// An identifier doiget recognises but cannot resolve yet (#500).
562///
563/// Only classes doiget can *name*. An entry carrying some field this does not
564/// know about still reports [`ParseError::NoIdentifier`], which stays correct
565/// for it: the point is not to guess, it is to stop saying "no identifier"
566/// about the cases where there demonstrably is one.
567///
568/// `pmid = {...}` is what PubMed's own BibTeX export writes. The BibLaTeX
569/// shape is `eprint = {...}` with `eprinttype = {pubmed}`, which
570/// [`arxiv_eligible`] already refuses -- correctly, and until now silently.
571fn unsupported_identifier(entry: &biblatex::Entry) -> Option<(&'static str, String)> {
572    let field = |name: &str| -> Option<String> {
573        let v = entry.get(name)?.format_verbatim().trim().to_string();
574        (!v.is_empty()).then_some(v)
575    };
576
577    if let Some(v) = field("pmid") {
578        return Some(("PMID", v));
579    }
580    if let Some(v) = field("pmcid") {
581        return Some(("PMCID", v));
582    }
583    let names_pubmed = entry
584        .get("archiveprefix")
585        .or_else(|| entry.get("eprinttype"))
586        .is_some_and(|c| c.format_verbatim().to_ascii_lowercase().contains("pubmed"));
587    if names_pubmed {
588        if let Some(v) = field("eprint") {
589            return Some(("PMID", v));
590        }
591    }
592    None
593}
594
595/// Whether an `eprint` field should be interpreted as an arXiv id.
596/// True when `archivePrefix` / `eprinttype` names arXiv (case-
597/// insensitive) or is absent; false when it explicitly names a
598/// different preprint server.
599fn arxiv_eligible(entry: &biblatex::Entry) -> bool {
600    match entry
601        .get("archiveprefix")
602        .or_else(|| entry.get("eprinttype"))
603    {
604        Some(chunks) => chunks
605            .format_verbatim()
606            .to_ascii_lowercase()
607            .contains("arxiv"),
608        None => true,
609    }
610}
611
612#[cfg(test)]
613#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
614mod tests {
615    /// #500, CSL-JSON. The BibTeX parser learned to distinguish "has no
616    /// identifier" from "has one I cannot use"; this format did not, so the
617    /// same PubMed record exported from Zotero still got the claim #500 exists
618    /// to prevent -- in the format a Zotero user is most likely to hand over.
619    #[test]
620    fn a_csl_entry_identified_only_by_a_pmid_is_not_missing_an_identifier() {
621        let doc = r#"[{"id":"Smith2020","type":"article-journal","PMID":"9659853"}]"#;
622        let out = parse_csl_json(doc);
623        assert_eq!(out.len(), 1);
624        match &out[0] {
625            Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => {
626                assert_eq!(*kind, "PMID");
627                assert_eq!(value, "9659853");
628            }
629            other => panic!("expected UnsupportedIdentifier, got {other:?}"),
630        }
631    }
632
633    /// Zotero does not emit a top-level `PMID`; it writes it into `note`.
634    /// A checker that only looked at the field would have missed the exporter
635    /// that produces most of these files.
636    #[test]
637    fn a_csl_note_carrying_a_pmid_is_read_the_same_way() {
638        let doc = r#"[{"id":"S","type":"article-journal","note":"PMID: 9659853; see PubMed"}]"#;
639        match &parse_csl_json(doc)[0] {
640            Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => {
641                assert_eq!(*kind, "PMID");
642                assert_eq!(value, "9659853");
643            }
644            other => panic!("expected UnsupportedIdentifier, got {other:?}"),
645        }
646    }
647
648    /// PMCID is the other half of the check and was written without a test.
649    /// Both the direct field and the Zotero `note` form.
650    #[test]
651    fn a_csl_entry_identified_only_by_a_pmcid_is_read_as_pmcid() {
652        for doc in [
653            r#"[{"id":"S","type":"article-journal","PMCID":"PMC1234567"}]"#,
654            r#"[{"id":"S","type":"article-journal","pmcid":"PMC1234567"}]"#,
655            r#"[{"id":"S","type":"article-journal","note":"PMCID: PMC1234567"}]"#,
656        ] {
657            match &parse_csl_json(doc)[0] {
658                Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => {
659                    assert_eq!(*kind, "PMCID", "for {doc}");
660                    assert_eq!(value, "PMC1234567", "for {doc}");
661                }
662                other => panic!("expected UnsupportedIdentifier for {doc}, got {other:?}"),
663            }
664        }
665    }
666
667    /// A numeric PMID must not be coerced into anything, and a DOI alongside
668    /// one still wins: the check runs only after every supported identifier
669    /// has been tried.
670    #[test]
671    fn a_csl_doi_still_wins_over_a_pmid() {
672        let doc = r#"[{"id":"S","type":"article-journal","DOI":"10.1234/x","PMID":9659853}]"#;
673        let parsed = parse_csl_json(doc);
674        let entry = parsed[0].as_ref().expect("the DOI must still win");
675        assert_eq!(entry.ref_.as_input_str(), "10.1234/x");
676    }
677
678    /// An entry with genuinely nothing must keep saying so -- the new check
679    /// must not turn every unidentifiable entry into "unsupported".
680    #[test]
681    fn a_csl_entry_with_no_identifier_at_all_still_says_so() {
682        let doc = r#"[{"id":"S","type":"article-journal","title":"No ids here"}]"#;
683        assert!(matches!(
684            &parse_csl_json(doc)[0],
685            Err(ParseError::NoIdentifier { .. })
686        ));
687    }
688
689    use super::*;
690
691    // ---- detect_format ---------------------------------------------
692
693    /// #500: the entry from PubMed's own BibTeX export. It carries `pmid`,
694    /// and reporting "entry has no DOI / arXiv id" about it is accurate about
695    /// the parser and wrong about the entry -- a user who believes it goes and
696    /// edits a bibliography that was fine.
697    ///
698    /// The PMID is real: `9659853` is Coryell 1998, whose DOI
699    /// `10.1176/ajp.155.7.895` NCBI's own esummary returns for it.
700    #[test]
701    fn a_pubmed_only_entry_says_it_has_a_pmid_not_that_it_has_nothing() {
702        let bib = r#"@article{coryell1998,
703  title = {Lithium discontinuation and subsequent effectiveness},
704  author = {Coryell, William},
705  year = {1998},
706  pmid = {9659853},
707}"#;
708        let out = parse_bibtex(bib);
709        assert_eq!(out.len(), 1);
710        match &out[0] {
711            Err(ParseError::UnsupportedIdentifier {
712                kind,
713                value,
714                entry_key,
715            }) => {
716                assert_eq!(kind, &"PMID");
717                assert_eq!(value, "9659853");
718                assert_eq!(entry_key.as_deref(), Some("coryell1998"));
719                let msg = out[0].as_ref().unwrap_err().to_string();
720                assert!(
721                    msg.contains("NOT missing an identifier"),
722                    "the message has to contradict the wrong conclusion explicitly, or the reader draws it anyway: {msg}"
723                );
724            }
725            other => panic!("expected UnsupportedIdentifier, got {other:?}"),
726        }
727    }
728
729    /// The BibLaTeX shape. `arxiv_eligible` already refused this -- correctly,
730    /// and until now silently, which is the whole complaint.
731    #[test]
732    fn the_biblatex_eprinttype_pubmed_shape_is_recognised_too() {
733        let bib = r#"@article{e,
734  title = {T},
735  eprint = {9659853},
736  eprinttype = {pubmed},
737}"#;
738        let out = parse_bibtex(bib);
739        assert!(
740            matches!(
741                &out[0],
742                Err(ParseError::UnsupportedIdentifier { kind: "PMID", .. })
743            ),
744            "got {:?}",
745            out[0]
746        );
747    }
748
749    /// An entry with genuinely nothing still reports `NoIdentifier`. The point
750    /// is not to relabel every failure -- it is to stop saying "no identifier"
751    /// about the cases where there demonstrably is one.
752    #[test]
753    fn an_entry_with_no_identifier_at_all_is_unchanged() {
754        let bib = "@article{x,
755  title = {T},
756  year = {2020},
757}";
758        let out = parse_bibtex(bib);
759        assert!(
760            matches!(&out[0], Err(ParseError::NoIdentifier { .. })),
761            "got {:?}",
762            out[0]
763        );
764    }
765
766    /// A DOI still wins. The new check runs only after every supported
767    /// identifier has been tried, so adding it cannot divert an entry doiget
768    /// could actually have resolved.
769    #[test]
770    fn a_doi_alongside_a_pmid_still_resolves() {
771        let bib = r#"@article{both,
772  title = {T},
773  doi = {10.1176/ajp.155.7.895},
774  pmid = {9659853},
775}"#;
776        let out = parse_bibtex(bib);
777        let parsed = out[0].as_ref().expect("the DOI must still win");
778        assert_eq!(parsed.ref_.as_input_str(), "10.1176/ajp.155.7.895");
779    }
780
781    #[test]
782    fn detect_by_bib_extension() {
783        let p = Utf8Path::new("/tmp/library.bib");
784        assert_eq!(detect_format(Some(p), ""), Format::Bibtex);
785    }
786
787    #[test]
788    fn detect_by_json_extension() {
789        let p = Utf8Path::new("/tmp/library.json");
790        assert_eq!(detect_format(Some(p), ""), Format::CslJson);
791    }
792
793    #[test]
794    fn detect_by_csl_extension() {
795        let p = Utf8Path::new("/tmp/library.csl");
796        assert_eq!(detect_format(Some(p), ""), Format::CslJson);
797    }
798
799    #[test]
800    fn detect_by_fingerprint_bibtex_at_sign() {
801        let body = "# comment\n\n@article{foo,\n  doi = {10.1/x}\n}\n";
802        assert_eq!(detect_format(None, body), Format::Bibtex);
803    }
804
805    #[test]
806    fn detect_by_fingerprint_csl_json_array() {
807        let body = "[{\"id\":\"foo\",\"DOI\":\"10.1/x\"}]";
808        assert_eq!(detect_format(None, body), Format::CslJson);
809    }
810
811    #[test]
812    fn detect_by_fingerprint_falls_through_to_refs() {
813        let body = "doi:10.1234/foo\narxiv:2401.12345\n";
814        assert_eq!(detect_format(None, body), Format::Refs);
815    }
816
817    // ---- plain refs ------------------------------------------------
818
819    #[test]
820    fn plain_refs_parses_mix_with_comments_and_blanks() {
821        let body = "\
822# header comment
823doi:10.1234/foo
824
825   arxiv:2401.12345
826# trailing comment
827";
828        let parsed = parse_plain_refs(body);
829        assert_eq!(parsed.len(), 2);
830        let okays: Vec<_> = parsed.into_iter().filter_map(Result::ok).collect();
831        assert!(matches!(okays[0].ref_, Ref::Doi(_)));
832        assert!(matches!(okays[1].ref_, Ref::Arxiv(_)));
833        assert!(okays.iter().all(|e| e.entry_key.is_none()));
834    }
835
836    #[test]
837    fn plain_refs_surface_per_line_invalid_refs() {
838        let body = "doi:10.1234/foo\nnot-a-ref\narxiv:2401.12345\n";
839        let parsed = parse_plain_refs(body);
840        assert_eq!(parsed.len(), 3);
841        assert!(parsed[0].is_ok());
842        assert!(matches!(parsed[1], Err(ParseError::InvalidRef { .. })));
843        assert!(parsed[2].is_ok());
844    }
845
846    // ---- CSL-JSON --------------------------------------------------
847
848    #[test]
849    fn csl_json_picks_doi_when_present() {
850        let body = r#"[{"id":"foo2024","DOI":"10.1234/foo"}]"#;
851        let parsed = parse_csl_json(body);
852        assert_eq!(parsed.len(), 1);
853        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
854        assert!(matches!(entry.ref_, Ref::Doi(_)));
855        assert_eq!(entry.entry_key.as_deref(), Some("foo2024"));
856    }
857
858    #[test]
859    fn csl_json_accepts_lowercase_doi_field() {
860        // Mendeley exports sometimes lowercase the field name.
861        let body = r#"[{"id":"x","doi":"10.5555/bar"}]"#;
862        let parsed = parse_csl_json(body);
863        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
864        assert!(matches!(entry.ref_, Ref::Doi(_)));
865    }
866
867    #[test]
868    fn csl_json_picks_arxiv_via_archive_prefix_and_eprint() {
869        let body = r#"[{"id":"arx","archivePrefix":"arXiv","eprint":"2401.12345"}]"#;
870        let parsed = parse_csl_json(body);
871        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
872        assert!(matches!(entry.ref_, Ref::Arxiv(_)));
873    }
874
875    #[test]
876    fn csl_json_arxiv_archive_prefix_is_case_insensitive() {
877        let body = r#"[{"id":"arx","archivePrefix":"ARXIV","eprint":"2401.12345"}]"#;
878        let parsed = parse_csl_json(body);
879        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
880        assert!(matches!(entry.ref_, Ref::Arxiv(_)));
881    }
882
883    #[test]
884    fn csl_json_doi_beats_arxiv_when_both_present() {
885        // ADR-0030 D3: priority is DOI > arXiv > PMID.
886        let body = r#"[{
887            "id":"both",
888            "DOI":"10.1234/foo",
889            "archivePrefix":"arXiv",
890            "eprint":"2401.12345"
891        }]"#;
892        let parsed = parse_csl_json(body);
893        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
894        assert!(matches!(entry.ref_, Ref::Doi(_)));
895    }
896
897    #[test]
898    fn csl_json_arxiv_from_note_field() {
899        // Zotero often dumps "arXiv:NNNN.NNNNN" into the note field
900        // instead of a typed field.
901        let body = r#"[{"id":"znote","note":"Comment: 12 pages. arXiv:2401.12345"}]"#;
902        let parsed = parse_csl_json(body);
903        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
904        assert!(matches!(entry.ref_, Ref::Arxiv(_)));
905    }
906
907    #[test]
908    fn csl_json_entry_without_any_identifier_yields_no_identifier_error() {
909        let body = r#"[{"id":"empty","title":"no ids here"}]"#;
910        let parsed = parse_csl_json(body);
911        assert!(matches!(
912            parsed.into_iter().next().unwrap(),
913            Err(ParseError::NoIdentifier { .. })
914        ));
915    }
916
917    #[test]
918    fn csl_json_invalid_doi_surface_as_invalid_ref_per_entry() {
919        let body = r#"[{"id":"bad","DOI":"not-a-doi"}]"#;
920        let parsed = parse_csl_json(body);
921        match &parsed[0] {
922            Err(ParseError::InvalidRef { raw, entry_key, .. }) => {
923                assert_eq!(raw, "not-a-doi");
924                assert_eq!(entry_key.as_deref(), Some("bad"));
925            }
926            other => panic!("expected InvalidRef, got {other:?}"),
927        }
928    }
929
930    #[test]
931    fn csl_json_top_level_malformed_yields_single_decode_error() {
932        let body = "{this is not JSON}";
933        let parsed = parse_csl_json(body);
934        assert_eq!(parsed.len(), 1);
935        assert!(matches!(
936            parsed[0],
937            Err(ParseError::Decode {
938                format: "csl-json",
939                ..
940            })
941        ));
942    }
943
944    #[test]
945    fn csl_json_non_array_top_level_yields_decode_error() {
946        // A single-entry object (not an array) is not a valid CSL-JSON
947        // document by the spec — the top level MUST be an array even
948        // for a single entry.
949        let body = r#"{"id":"x","DOI":"10.1/x"}"#;
950        let parsed = parse_csl_json(body);
951        assert!(matches!(
952            parsed[0],
953            Err(ParseError::Decode {
954                format: "csl-json",
955                ..
956            })
957        ));
958    }
959
960    // ---- parse_input dispatch -------------------------------------
961
962    #[test]
963    fn parse_input_auto_dispatches_csl_json_by_content() {
964        let body = r#"[{"id":"foo","DOI":"10.1234/foo"}]"#;
965        let parsed = parse_input(body, Format::Auto, None);
966        assert_eq!(parsed.len(), 1);
967        assert!(matches!(
968            parsed[0],
969            Ok(ParsedEntry {
970                ref_: Ref::Doi(_),
971                ..
972            })
973        ));
974    }
975
976    #[test]
977    fn parse_input_auto_dispatches_refs_by_content() {
978        let body = "doi:10.1234/foo\n";
979        let parsed = parse_input(body, Format::Auto, None);
980        assert_eq!(parsed.len(), 1);
981        assert!(matches!(
982            parsed[0],
983            Ok(ParsedEntry {
984                ref_: Ref::Doi(_),
985                ..
986            })
987        ));
988    }
989
990    // ---- BibTeX parsing (ADR-0030 D2) -----------------------------
991
992    #[test]
993    fn bibtex_picks_doi_and_preserves_key() {
994        let body = r#"@article{Onsager1944,
995            author = {Onsager, Lars},
996            title  = {Crystal Statistics},
997            doi    = {10.1103/PhysRev.65.117}
998        }"#;
999        let parsed = parse_bibtex(body);
1000        assert_eq!(parsed.len(), 1);
1001        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
1002        assert!(matches!(entry.ref_, Ref::Doi(_)));
1003        assert_eq!(entry.entry_key.as_deref(), Some("Onsager1944"));
1004    }
1005
1006    #[test]
1007    fn bibtex_picks_arxiv_via_eprint() {
1008        let body = r#"@article{Pollmann2012,
1009            title         = {Detection of SPT order},
1010            eprint        = {1010.3732},
1011            archivePrefix = {arXiv}
1012        }"#;
1013        let entry = parse_bibtex(body)
1014            .into_iter()
1015            .next()
1016            .unwrap()
1017            .expect("entry parses");
1018        assert!(matches!(entry.ref_, Ref::Arxiv(_)));
1019    }
1020
1021    #[test]
1022    fn bibtex_bare_eprint_without_prefix_is_arxiv() {
1023        // The dominant convention: a lone `eprint` field is an arXiv id.
1024        let body = "@misc{x, eprint = {2204.12345}}";
1025        let entry = parse_bibtex(body)
1026            .into_iter()
1027            .next()
1028            .unwrap()
1029            .expect("entry parses");
1030        assert!(matches!(entry.ref_, Ref::Arxiv(_)));
1031    }
1032
1033    #[test]
1034    fn bibtex_non_arxiv_eprinttype_reports_the_identifier_it_found() {
1035        // This test used to assert `NoIdentifier`, with the comment "the entry
1036        // has no resolvable identifier". The entry has a PMID. It is not
1037        // resolvable BY DOIGET, which is a different statement, and the one
1038        // #500 is about -- so the test was pinning the wrong claim.
1039        let body = "@article{x, eprint = {12345678}, eprinttype = {pubmed}}";
1040        let res = parse_bibtex(body).into_iter().next().unwrap();
1041        match res {
1042            Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => {
1043                assert_eq!(kind, "PMID");
1044                assert_eq!(value, "12345678");
1045            }
1046            other => panic!("expected UnsupportedIdentifier, got {other:?}"),
1047        }
1048    }
1049
1050    #[test]
1051    fn bibtex_doi_beats_arxiv_when_both_present() {
1052        // ADR-0030 D3: priority is DOI > arXiv > PMID.
1053        let body = r#"@article{both,
1054            doi           = {10.1234/foo},
1055            eprint        = {2401.12345},
1056            archivePrefix = {arXiv}
1057        }"#;
1058        let entry = parse_bibtex(body)
1059            .into_iter()
1060            .next()
1061            .unwrap()
1062            .expect("entry parses");
1063        assert!(matches!(entry.ref_, Ref::Doi(_)));
1064    }
1065
1066    #[test]
1067    fn bibtex_multiple_entries_each_yield_a_result() {
1068        let body = r#"
1069            @article{a, doi = {10.1103/PhysRev.65.117}}
1070            @article{b, eprint = {1010.3732}, archivePrefix = {arXiv}}
1071            @article{c, title = {no identifier here}}
1072        "#;
1073        let parsed = parse_bibtex(body);
1074        assert_eq!(parsed.len(), 3);
1075        assert!(matches!(
1076            parsed[0],
1077            Ok(ParsedEntry {
1078                ref_: Ref::Doi(_),
1079                ..
1080            })
1081        ));
1082        assert!(matches!(
1083            parsed[1],
1084            Ok(ParsedEntry {
1085                ref_: Ref::Arxiv(_),
1086                ..
1087            })
1088        ));
1089        assert!(matches!(parsed[2], Err(ParseError::NoIdentifier { .. })));
1090    }
1091
1092    #[test]
1093    fn bibtex_entry_without_identifier_yields_no_identifier_error() {
1094        let body = "@book{nodoi, title = {A Book}, author = {Author, A.}}";
1095        let res = parse_bibtex(body).into_iter().next().unwrap();
1096        assert!(matches!(res, Err(ParseError::NoIdentifier { .. })));
1097    }
1098
1099    #[test]
1100    fn bibtex_invalid_doi_surfaces_as_invalid_ref_per_entry() {
1101        let body = "@article{bad, doi = {not-a-doi}}";
1102        let res = parse_bibtex(body).into_iter().next().unwrap();
1103        assert!(matches!(res, Err(ParseError::InvalidRef { .. })));
1104    }
1105
1106    #[test]
1107    fn bibtex_malformed_input_yields_single_decode_error() {
1108        // A truncated entry the biblatex parser rejects outright.
1109        let body = "@article{unterminated, doi = {10.1234/x}";
1110        let parsed = parse_bibtex(body);
1111        assert_eq!(parsed.len(), 1);
1112        assert!(matches!(
1113            parsed[0],
1114            Err(ParseError::Decode {
1115                format: "bibtex",
1116                ..
1117            })
1118        ));
1119    }
1120
1121    #[test]
1122    fn parse_input_bibtex_dispatches_and_parses() {
1123        let body = "@article{foo, doi = {10.1234/foo}}";
1124        let parsed = parse_input(body, Format::Bibtex, None);
1125        assert_eq!(parsed.len(), 1);
1126        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
1127        assert!(matches!(entry.ref_, Ref::Doi(_)));
1128        assert_eq!(entry.entry_key.as_deref(), Some("foo"));
1129    }
1130
1131    #[test]
1132    fn parse_input_auto_dispatches_bibtex_by_content() {
1133        // Leading `@article{` fingerprint routes to the BibTeX parser.
1134        let body = "@article{auto, doi = {10.1234/auto}}";
1135        let parsed = parse_input(body, Format::Auto, None);
1136        let entry = parsed.into_iter().next().unwrap().expect("entry parses");
1137        assert!(matches!(entry.ref_, Ref::Doi(_)));
1138    }
1139
1140    #[test]
1141    fn parse_input_auto_with_path_uses_extension() {
1142        let body = "[]";
1143        let parsed = parse_input(body, Format::Auto, Some(Utf8Path::new("foo.csl")));
1144        assert_eq!(
1145            parsed.len(),
1146            0,
1147            "empty array yields zero entries: {parsed:?}"
1148        );
1149    }
1150
1151    // ---- Format::as_wire ------------------------------------------
1152
1153    #[test]
1154    fn format_wire_strings_are_stable() {
1155        // Pinned because the strings appear in the CLI --format flag,
1156        // the MCP tool input schema, and the JSON-Lines parse-error
1157        // records (ADR-0030 §6). A drift would be a wire-format break.
1158        assert_eq!(Format::Auto.as_wire(), "auto");
1159        assert_eq!(Format::Refs.as_wire(), "refs");
1160        assert_eq!(Format::CslJson.as_wire(), "csl-json");
1161        assert_eq!(Format::Bibtex.as_wire(), "bibtex");
1162    }
1163
1164    #[test]
1165    fn the_unsupported_identifier_claim_denies_the_wrong_reading() {
1166        // #500's whole point: the sentence must put the gap on doiget's
1167        // side. A reader who takes "no identifier" at face value goes and
1168        // edits a `.bib` that was fine.
1169        let msg = unsupported_identifier_claim("PMID", "9659853");
1170        assert!(msg.contains("PMID"), "names the identifier kind: {msg}");
1171        assert!(msg.contains("9659853"), "quotes the value: {msg}");
1172        assert!(
1173            msg.contains("NOT missing an identifier"),
1174            "denies the wrong reading: {msg}"
1175        );
1176        assert!(msg.contains("#500"), "points at the issue: {msg}");
1177        // The `entry_key` prefix belongs to `Display`, not here -- callers
1178        // carry it in a field of its own and would say it twice.
1179        assert!(!msg.starts_with("entry {"), "no entry_key prefix: {msg}");
1180    }
1181
1182    /// #614: an entry with no DOI or arXiv id whose url is a GitHub
1183    /// repository or release is software, in both formats; any other URL
1184    /// leaves the entry id-less.
1185    #[test]
1186    fn a_github_url_makes_an_id_less_entry_software() {
1187        let bib = parse_bibtex(
1188            "@software{hf, title={HFDMRG}, url={https://github.com/srwhite59/HFDMRG.jl/releases/tag/v0.1.0}}\n\
1189             @misc{web, title={A page}, url={https://example.org/page}}\n",
1190        );
1191        assert_eq!(
1192            bib[0],
1193            Err(ParseError::SoftwareUrl {
1194                url: "https://github.com/srwhite59/HFDMRG.jl/releases/tag/v0.1.0".into(),
1195                entry_key: Some("hf".into()),
1196            })
1197        );
1198        assert!(matches!(bib[1], Err(ParseError::NoIdentifier { .. })));
1199        let csl = parse_csl_json(
1200            r#"[{"id":"hf","type":"software","title":"HFDMRG","URL":"https://github.com/srwhite59/HFDMRG.jl"}]"#,
1201        );
1202        assert!(
1203            matches!(&csl[0], Err(ParseError::SoftwareUrl { url, .. }) if url.ends_with("HFDMRG.jl"))
1204        );
1205    }
1206}