Skip to main content

doiget_core/
markup.rs

1//! Plain text from publisher metadata strings that carry inline markup.
2//!
3//! Crossref serves titles as the publisher deposited them, and several
4//! publishers deposit JATS / MathML inline elements pretty-printed onto their
5//! own lines (#609):
6//!
7//! ```text
8//! "Recent developments in the P\n                    <scp>y</scp>\n                    SCF program package"
9//! ```
10//!
11//! Stripping the tags is not enough. The newline-plus-indent runs around an
12//! inline element were added by the pretty-printer, so they say nothing about
13//! whether the author wrote a space there: `P<scp>y</scp>SCF` is one word,
14//! `in <i>γ</i>-InSe` is two. [`plain_title`] decides each such run from its
15//! neighbours; the rules are pinned against real deposits in the tests below.
16//!
17//! Whitespace that is not a pretty-printer run (no line break, or no tag
18//! beside it) collapses to one space, and `$...$` is left alone: a
19//! hand-written TeX title is content, which `doiget lint` already treats as
20//! intentional.
21
22/// `raw` with inline markup reduced to its text content, XML entities
23/// decoded and whitespace collapsed. A string with no tag, entity or line
24/// break is returned unchanged.
25#[must_use]
26pub fn plain_title(raw: &str) -> String {
27    if !raw.contains(['<', '&', '\n', '\r']) {
28        return raw.to_string();
29    }
30    let mut out = String::with_capacity(raw.len());
31    // The previous non-blank text run, and what lies between it and the next.
32    let mut prev: Option<(&str, usize)> = None;
33    let mut gap = Gap::default();
34    for chunk in split_on_tags(raw) {
35        let (text, depth) = match chunk {
36            Chunk::Tag => {
37                gap.tag = true;
38                continue;
39            }
40            Chunk::Text { text, depth } => (text, depth),
41        };
42        let body = text.trim();
43        if body.is_empty() {
44            gap.absorb(text);
45            continue;
46        }
47        gap.absorb(&text[..text.len() - text.trim_start().len()]);
48        if let Some((prev_body, prev_depth)) = prev {
49            out.push_str(if gap.tag && gap.line_break {
50                boundary(prev_body, prev_depth, body, depth)
51            } else if gap.space {
52                " "
53            } else {
54                ""
55            });
56        }
57        out.push_str(&collapse(body));
58        prev = Some((body, depth));
59        gap = Gap::default();
60        gap.absorb(&text[text.trim_end().len()..]);
61    }
62    decode_entities(&out)
63}
64
65/// Whether `s` carries a line break or an inline markup tag, i.e. whether
66/// [`plain_title`] would rewrite more than entities. `doiget lint` uses this
67/// to find entries pasted from an older `doiget cite` (#609).
68#[must_use]
69pub fn has_inline_markup(s: &str) -> bool {
70    s.contains(['\n', '\r']) || split_on_tags(s).iter().any(|c| matches!(c, Chunk::Tag))
71}
72
73/// What separates two text runs.
74#[derive(Default)]
75struct Gap {
76    tag: bool,
77    space: bool,
78    line_break: bool,
79}
80
81impl Gap {
82    fn absorb(&mut self, ws: &str) {
83        self.space |= !ws.is_empty();
84        self.line_break |= ws.contains(['\n', '\r']);
85    }
86}
87
88enum Chunk<'a> {
89    Tag,
90    /// Text between tags, and how many inline elements enclose it.
91    Text {
92        text: &'a str,
93        depth: usize,
94    },
95}
96
97/// Split into text runs and tags. A `<` that does not open a tag (`a < b`)
98/// stays text.
99fn split_on_tags(raw: &str) -> Vec<Chunk<'_>> {
100    let mut chunks = Vec::new();
101    let mut depth = 0usize;
102    let mut start = 0;
103    let mut i = 0;
104    // `<` is ASCII, so every index this slices at is a char boundary.
105    while let Some(off) = raw[i..].find('<') {
106        let at = i + off;
107        match tag_len(&raw[at..]) {
108            Some(len) => {
109                if start < at {
110                    chunks.push(Chunk::Text {
111                        text: &raw[start..at],
112                        depth,
113                    });
114                }
115                let tag = &raw[at..at + len];
116                if tag.starts_with("</") {
117                    depth = depth.saturating_sub(1);
118                } else if !tag.ends_with("/>") {
119                    depth += 1;
120                }
121                chunks.push(Chunk::Tag);
122                i = at + len;
123                start = i;
124            }
125            None => i = at + 1,
126        }
127    }
128    if start < raw.len() {
129        chunks.push(Chunk::Text {
130            text: &raw[start..],
131            depth,
132        });
133    }
134    chunks
135}
136
137/// Inline elements publishers deposit in titles. An opening tag with one of
138/// these names is markup even when its closing tag is elsewhere; any other
139/// name counts only if a matching `</name` follows (see [`tag_len`]).
140const INLINE_ELEMENTS: &[&str] = &[
141    "b",
142    "bold",
143    "br",
144    "em",
145    "i",
146    "inline-formula",
147    "italic",
148    "math",
149    "monospace",
150    "sc",
151    "scp",
152    "small",
153    "span",
154    "strong",
155    "sub",
156    "sup",
157    "tex-math",
158    "tt",
159    "u",
160    "underline",
161];
162
163/// Length of the tag at the start of `s`, if it is one.
164///
165/// A real grammar rather than "`<` up to the next `>`": the name is
166/// `[A-Za-z][A-Za-z0-9:._-]*` and anything before `>` must be
167/// `name="value"` attributes. The loose rule treated `T<Tc in samples with
168/// applied field H>Hc2` as one tag and stored `THc2` (review of #618). An
169/// opening tag must also be plausible as markup: a known inline element, a
170/// namespaced one (`mml:mi`, `jats:italic`), or one whose `</name` follows --
171/// so `x<y>z` stays text.
172fn tag_len(s: &str) -> Option<usize> {
173    let bytes = s.as_bytes();
174    let mut i = 1; // past '<'
175    let closing = bytes.get(i) == Some(&b'/');
176    if closing {
177        i += 1;
178    }
179    let name_start = i;
180    if !bytes.get(i)?.is_ascii_alphabetic() {
181        return None;
182    }
183    while bytes
184        .get(i)
185        .is_some_and(|b| b.is_ascii_alphanumeric() || matches!(b, b':' | b'.' | b'_' | b'-'))
186    {
187        i += 1;
188    }
189    let name = &s[name_start..i];
190    let mut self_closing = false;
191    loop {
192        while bytes.get(i).is_some_and(u8::is_ascii_whitespace) {
193            i += 1;
194        }
195        match bytes.get(i)? {
196            b'>' => {
197                i += 1;
198                break;
199            }
200            b'/' if !closing && bytes.get(i + 1) == Some(&b'>') => {
201                self_closing = true;
202                i += 2;
203                break;
204            }
205            b if !closing && (b.is_ascii_alphabetic() || *b == b'_') => {
206                // attribute: name = "value" | 'value'
207                while bytes.get(i).is_some_and(|b| {
208                    b.is_ascii_alphanumeric() || matches!(b, b':' | b'.' | b'_' | b'-')
209                }) {
210                    i += 1;
211                }
212                while bytes.get(i).is_some_and(u8::is_ascii_whitespace) {
213                    i += 1;
214                }
215                if bytes.get(i) != Some(&b'=') {
216                    return None;
217                }
218                i += 1;
219                while bytes.get(i).is_some_and(u8::is_ascii_whitespace) {
220                    i += 1;
221                }
222                let quote = *bytes.get(i)?;
223                if !matches!(quote, b'"' | b'\'') {
224                    return None;
225                }
226                let close = s[i + 1..].find(quote as char)?;
227                if s[i + 1..i + 1 + close].contains('<') {
228                    return None;
229                }
230                i += close + 2;
231            }
232            _ => return None,
233        }
234    }
235    let known = INLINE_ELEMENTS.contains(&name.to_ascii_lowercase().as_str()) || name.contains(':');
236    let plausible = closing || self_closing || known || s[i..].contains(&format!("</{name}"));
237    plausible.then_some(i)
238}
239
240fn collapse(s: &str) -> String {
241    s.split_whitespace().collect::<Vec<_>>().join(" ")
242}
243
244/// The separator for a pretty-printer run between `left` and `right` that
245/// crosses a tag.
246///
247/// The side enclosed by more elements is the element's content. A word
248/// inside it (`<i>Escherichia coli</i>`, `<b>31</b>`) is spaced from a
249/// neighbouring word. A symbol (`<i>x</i>`, `<scp>y</scp>`, `<i>β</i>`) is
250/// joined to its neighbour unless that neighbour is a lower-case word,
251/// sentence punctuation before it, or a relational operator.
252fn boundary(left: &str, left_depth: usize, right: &str, right_depth: usize) -> &'static str {
253    if left_depth == right_depth {
254        // `<i>L</i>\n<i>p</i>`: two elements and nothing outside to go by.
255        return "";
256    }
257    let (inner, outer, outer_is_left) = if right_depth > left_depth {
258        (right, left, true)
259    } else {
260        (left, right, false)
261    };
262    let edge = if outer_is_left {
263        outer.chars().next_back()
264    } else {
265        outer.chars().next()
266    };
267    let Some(edge) = edge else { return "" };
268    let is_word = inner.chars().filter(|c| c.is_alphanumeric()).count() >= 2;
269    // `&` is the start of an escaped `&gt;` / `&lt;`, decoded after this.
270    let operator = matches!(edge, '=' | '<' | '>' | '≤' | '≥' | '±' | '&');
271    let sentence_punct = outer_is_left && matches!(edge, '.' | ',' | ';' | ':');
272    let space = if is_word {
273        edge.is_alphanumeric() || sentence_punct
274    } else {
275        lowercase_word_at_edge(outer, outer_is_left) || sentence_punct || operator
276    };
277    if space {
278        " "
279    } else {
280        ""
281    }
282}
283
284/// Whether the letters at the inner edge of `outer` form a lower-case word
285/// of two or more letters (`in`, `noise`, the `grown` of `MOVPE-grown`).
286fn lowercase_word_at_edge(outer: &str, at_end: bool) -> bool {
287    let is_lower_word =
288        |letters: &[char]| letters.len() >= 2 && letters.iter().all(|c| c.is_lowercase());
289    if at_end {
290        let letters: Vec<char> = outer
291            .chars()
292            .rev()
293            .take_while(|c| c.is_alphabetic())
294            .collect();
295        is_lower_word(&letters)
296    } else {
297        let letters: Vec<char> = outer.chars().take_while(|c| c.is_alphabetic()).collect();
298        is_lower_word(&letters)
299    }
300}
301
302/// Decode the XML entities a deposit carries. Crossref sometimes escapes a
303/// deposit's own entity a second time (`&amp;gt;` for `>`), so a string that
304/// contained `&amp;` is decoded twice.
305fn decode_entities(s: &str) -> String {
306    if !s.contains('&') {
307        return s.to_string();
308    }
309    let once = decode_once(s);
310    if s.contains("&amp;") && once.contains('&') {
311        decode_once(&once)
312    } else {
313        once
314    }
315}
316
317fn decode_once(s: &str) -> String {
318    let mut out = String::with_capacity(s.len());
319    let mut rest = s;
320    while let Some(pos) = rest.find('&') {
321        out.push_str(&rest[..pos]);
322        let tail = &rest[pos..];
323        match entity(tail) {
324            Some((c, len)) => {
325                out.push(c);
326                rest = &tail[len..];
327            }
328            None => {
329                out.push('&');
330                rest = &tail[1..];
331            }
332        }
333    }
334    out.push_str(rest);
335    out
336}
337
338/// The character and byte length of the entity at the start of `s`.
339fn entity(s: &str) -> Option<(char, usize)> {
340    let end = s.find(';').filter(|&end| end <= 10)?;
341    let name = &s[1..end];
342    let c = match name {
343        "amp" => '&',
344        "lt" => '<',
345        "gt" => '>',
346        "quot" => '"',
347        "apos" => '\'',
348        _ => {
349            let code = match name.strip_prefix("#x").or_else(|| name.strip_prefix("#X")) {
350                Some(hex) => u32::from_str_radix(hex, 16).ok()?,
351                None => name.strip_prefix('#')?.parse().ok()?,
352            };
353            char::from_u32(code)?
354        }
355    };
356    Some((c, end + 1))
357}
358
359#[cfg(test)]
360mod tests {
361    use super::plain_title;
362
363    /// Real Crossref deposits (AIP, prefix `10.1063`, sampled 2026-09-29),
364    /// shortened, with the pretty-printer's indentation kept verbatim.
365    #[test]
366    fn pretty_printed_inline_markup_reads_as_the_title_the_author_wrote() {
367        let ind = "\n                    ";
368        let cases = [
369            (
370                format!("Recent developments in the P{ind}<scp>y</scp>{ind}SCF program package"),
371                "Recent developments in the PySCF program package",
372            ),
373            (
374                format!("control of [Pr1−{ind}<i>x</i>{ind}Ca{ind}<i>x</i>{ind}MnO3/SrTiO3]15 superlattices"),
375                "control of [Pr1−xCaxMnO3/SrTiO3]15 superlattices",
376            ),
377            (
378                format!("Thermal transport in{ind}<b>{ind}  <i>γ</i>{ind}</b>{ind}-InSe: Bulk single crystals"),
379                "Thermal transport in γ-InSe: Bulk single crystals",
380            ),
381            (
382                format!("Erratum: [Phys. Plasmas{ind}<b>31</b>{ind}, 042509 (2024)]"),
383                "Erratum: [Phys. Plasmas 31, 042509 (2024)]",
384            ),
385            (
386                format!("[Appl. Phys. Lett.{ind}<b>117</b>{ind}, 092101 (2020)]"),
387                "[Appl. Phys. Lett. 117, 092101 (2020)]",
388            ),
389            (format!("1/{ind}<i>f</i>{ind}noise model"), "1/f noise model"),
390            (
391                format!("Approximate Symmetry in{ind}<i>Z’</i>{ind}&amp;gt; 1 Structures"),
392                "Approximate Symmetry in Z’ > 1 Structures",
393            ),
394            (
395                format!("coupling in 2{ind}<i>H</i>{ind}-VSe2 bilayer"),
396                "coupling in 2H-VSe2 bilayer",
397            ),
398            (
399                format!("their isomers{ind}<i>cis</i>{ind}- and{ind}<i>trans</i>{ind}-HNCHO"),
400                "their isomers cis- and trans-HNCHO",
401            ),
402            (
403                format!("insights into{ind}<i>Escherichia coli</i>{ind}O32:H37 contact"),
404                "insights into Escherichia coli O32:H37 contact",
405            ),
406            (
407                format!("non-commutative{ind}<i>L</i>{ind}<i>p</i>{ind}-spaces"),
408                "non-commutative Lp-spaces",
409            ),
410            (
411                format!("MOVPE-grown <b>{ind} <i>β</i>{ind}</b>-Ga2O3 films"),
412                "MOVPE-grown β-Ga2O3 films",
413            ),
414            (
415                format!("in Ce2(Cu1<b>−</b>{ind}<i>x</i>Ni<i>x</i>)2In"),
416                "in Ce2(Cu1−xNix)2In",
417            ),
418            (
419                format!("operators with <i>L</i>{ind}<i>p</i> potentials"),
420                "operators with Lp potentials",
421            ),
422        ];
423        for (raw, want) in cases {
424            assert_eq!(plain_title(&raw), want, "raw: {raw:?}");
425        }
426    }
427
428    /// Review of #618: a letter after `<` is an inequality in a physics
429    /// title far more often than a tag, and the text up to some later `>`
430    /// must never be taken for one.
431    #[test]
432    fn an_inequality_is_not_mistaken_for_a_tag() {
433        for s in [
434            "Resistivity anomaly for T<Tc in samples with applied field H>Hc2",
435            "Comparing groups where n<N states and m>M bands coexist",
436            "a<b and c>d",
437            "the regime x<y>z",
438            "for L<M, the <unclosed",
439        ] {
440            assert_eq!(plain_title(s), s);
441            assert!(!super::has_inline_markup(s), "{s}");
442        }
443        // Real markup next to an inequality is still reduced.
444        assert_eq!(
445            plain_title("T<Tc for <i>x</i> > 0 and <span class=\"x\">y</span>"),
446            "T<Tc for x > 0 and y"
447        );
448        assert_eq!(plain_title("a <jats:italic>b</jats:italic> c"), "a b c");
449        assert_eq!(plain_title("an <unknown-el>x</unknown-el> y"), "an x y");
450    }
451
452    #[test]
453    fn plain_strings_and_tex_titles_pass_through_unchanged() {
454        for s in [
455            "Density matrix formulation for quantum renormalization groups",
456            "The $T\\bar{T}$ deformation",
457            "  two  spaces  ",
458        ] {
459            assert_eq!(plain_title(s), s);
460        }
461        assert_eq!(plain_title("a < b and c > d"), "a < b and c > d");
462    }
463
464    #[test]
465    fn ordinary_spacing_around_inline_elements_is_kept() {
466        assert_eq!(
467            plain_title("The <i>ab initio</i> method for H<sub>2</sub>O"),
468            "The ab initio method for H2O"
469        );
470        assert_eq!(plain_title("Line one\n  line two"), "Line one line two");
471    }
472
473    #[test]
474    fn entities_decode_and_a_bare_ampersand_survives() {
475        assert_eq!(
476            plain_title("Tom &amp; Jerry &#x3B2; &#946;"),
477            "Tom & Jerry β β"
478        );
479        assert_eq!(plain_title("R&D &unknown; &"), "R&D &unknown; &");
480    }
481
482    #[test]
483    fn markup_detection_ignores_bare_comparisons_and_tex() {
484        use super::has_inline_markup;
485        assert!(has_inline_markup(
486            "the P\n    <scp>y</scp>\n    SCF package"
487        ));
488        assert!(has_inline_markup("Spin-<i>S</i> chains"));
489        assert!(has_inline_markup("Line one\nline two"));
490        assert!(!has_inline_markup("Regime a < b holds"));
491        assert!(!has_inline_markup("The $T\\bar{T}$ deformation"));
492    }
493
494    #[test]
495    fn mathml_reduces_to_its_text_content() {
496        assert_eq!(
497            plain_title(
498                "Spin <mml:math><mml:msub><mml:mi>S</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math> chains"
499            ),
500            "Spin S1 chains"
501        );
502    }
503}