Skip to main content

koan_core/remote/
wikimedia.rs

1//! Wikidata, Wikipedia and Wikimedia Commons — where an artist's biography and
2//! photograph come from.
3//!
4//! Reached through the artist's Wikidata item, which MusicBrainz links to. No
5//! API key; Wikimedia asks only for a User-Agent that says who is calling.
6
7use serde::Deserialize;
8use thiserror::Error;
9
10const USER_AGENT: &str = concat!(
11    "koan/",
12    env!("CARGO_PKG_VERSION"),
13    " (https://github.com/radiosilence/koan)"
14);
15
16/// The article language. English has by far the widest coverage of musicians.
17const WIKI: &str = "enwiki";
18const WIKIPEDIA_API: &str = "https://en.wikipedia.org/w/api.php";
19const COMMONS_API: &str = "https://commons.wikimedia.org/w/api.php";
20
21#[derive(Debug, Error)]
22pub enum WikimediaError {
23    #[error("http error: {0}")]
24    Http(#[from] reqwest::Error),
25}
26
27/// What a Wikidata item says about where to read more and what the artist
28/// looks like.
29#[derive(Debug, Default, PartialEq)]
30pub struct Entity {
31    /// The English Wikipedia article's title.
32    pub article: Option<String>,
33    /// A Commons file name (P18, "image").
34    pub image: Option<String>,
35}
36
37/// An article's lead section, as plain text.
38#[derive(Debug, PartialEq)]
39pub struct Intro {
40    pub text: String,
41    pub url: String,
42}
43
44/// A Commons image, resized, with the credit its licence asks for.
45#[derive(Debug, PartialEq)]
46pub struct Image {
47    pub url: String,
48    pub credit: Option<String>,
49}
50
51pub fn client() -> reqwest::blocking::Client {
52    reqwest::blocking::Client::builder()
53        .user_agent(USER_AGENT)
54        .timeout(std::time::Duration::from_secs(15))
55        .build()
56        .unwrap_or_else(|_| reqwest::blocking::Client::new())
57}
58
59pub fn entity(http: &reqwest::blocking::Client, qid: &str) -> Result<Entity, WikimediaError> {
60    let url = format!("https://www.wikidata.org/wiki/Special:EntityData/{qid}.json");
61    let body: serde_json::Value = http.get(url).send()?.error_for_status()?.json()?;
62    Ok(parse_entity(&body, qid))
63}
64
65fn parse_entity(body: &serde_json::Value, qid: &str) -> Entity {
66    // A redirected item comes back under its new id, so take whichever is there.
67    let item = body["entities"]
68        .get(qid)
69        .or_else(|| body["entities"].as_object()?.values().next());
70    let Some(item) = item else {
71        return Entity::default();
72    };
73    Entity {
74        article: item["sitelinks"][WIKI]["title"].as_str().map(String::from),
75        image: item["claims"]["P18"][0]["mainsnak"]["datavalue"]["value"]
76            .as_str()
77            .map(String::from),
78    }
79}
80
81pub fn intro(
82    http: &reqwest::blocking::Client,
83    article: &str,
84) -> Result<Option<Intro>, WikimediaError> {
85    #[derive(Deserialize)]
86    struct Response {
87        query: Query,
88    }
89    #[derive(Deserialize)]
90    struct Query {
91        pages: Vec<Page>,
92    }
93    #[derive(Deserialize)]
94    struct Page {
95        extract: Option<String>,
96        fullurl: Option<String>,
97    }
98
99    let response: Response = http
100        .get(WIKIPEDIA_API)
101        .query(&[
102            ("action", "query"),
103            ("prop", "extracts|info"),
104            ("inprop", "url"),
105            ("exintro", "1"),
106            ("explaintext", "1"),
107            ("redirects", "1"),
108            ("format", "json"),
109            ("formatversion", "2"),
110            ("titles", article),
111        ])
112        .send()?
113        .error_for_status()?
114        .json()?;
115    Ok(response.query.pages.into_iter().next().and_then(|page| {
116        let text = page.extract?.trim().to_string();
117        (!text.is_empty()).then(|| Intro {
118            text,
119            url: page.fullurl.unwrap_or_default(),
120        })
121    }))
122}
123
124pub fn image(
125    http: &reqwest::blocking::Client,
126    file: &str,
127    width: u32,
128) -> Result<Option<Image>, WikimediaError> {
129    let body: serde_json::Value = http
130        .get(COMMONS_API)
131        .query(&[
132            ("action", "query"),
133            ("prop", "imageinfo"),
134            ("iiprop", "url|extmetadata"),
135            ("iiextmetadatafilter", "Artist|LicenseShortName"),
136            ("iiurlwidth", &width.to_string()),
137            ("format", "json"),
138            ("formatversion", "2"),
139            ("titles", &format!("File:{file}")),
140        ])
141        .send()?
142        .error_for_status()?
143        .json()?;
144    Ok(parse_image(&body))
145}
146
147fn parse_image(body: &serde_json::Value) -> Option<Image> {
148    let info = &body["query"]["pages"][0]["imageinfo"][0];
149    let url = info["thumburl"].as_str().or(info["url"].as_str())?;
150    let meta = &info["extmetadata"];
151    let author = meta["Artist"]["value"]
152        .as_str()
153        .map(strip_tags)
154        .filter(|s| !s.is_empty());
155    let licence = meta["LicenseShortName"]["value"]
156        .as_str()
157        .map(strip_tags)
158        .filter(|s| !s.is_empty());
159    let credit = match (author, licence) {
160        (Some(author), Some(licence)) => Some(format!("{author} · {licence}")),
161        (author, licence) => author.or(licence),
162    };
163    Some(Image {
164        url: url.to_string(),
165        credit,
166    })
167}
168
169pub fn download(http: &reqwest::blocking::Client, url: &str) -> Result<Vec<u8>, WikimediaError> {
170    Ok(http.get(url).send()?.error_for_status()?.bytes()?.to_vec())
171}
172
173/// Commons credits are HTML — usually a link to the photographer's profile.
174fn strip_tags(html: &str) -> String {
175    let mut text = String::with_capacity(html.len());
176    let mut in_tag = false;
177    for c in html.chars() {
178        match c {
179            '<' => in_tag = true,
180            '>' => in_tag = false,
181            c if !in_tag => text.push(c),
182            _ => {}
183        }
184    }
185    text.replace("&amp;", "&")
186        .replace("&quot;", "\"")
187        .replace("&#039;", "'")
188        .split_whitespace()
189        .collect::<Vec<_>>()
190        .join(" ")
191}
192
193#[cfg(test)]
194mod tests {
195    use super::*;
196    use serde_json::json;
197
198    #[test]
199    fn entity_reads_the_article_and_the_image() {
200        let body = json!({"entities": {"Q2358013": {
201            "sitelinks": {"enwiki": {"title": "Glass Candy"}},
202            "claims": {"P18": [{"mainsnak": {"datavalue": {
203                "value": "Glass Candy Ida No and Johnny Jewel.jpg"
204            }}}]}
205        }}});
206        assert_eq!(
207            parse_entity(&body, "Q2358013"),
208            Entity {
209                article: Some("Glass Candy".into()),
210                image: Some("Glass Candy Ida No and Johnny Jewel.jpg".into()),
211            }
212        );
213    }
214
215    #[test]
216    fn entity_follows_a_redirected_item() {
217        let body = json!({"entities": {"Q2": {"sitelinks": {"enwiki": {"title": "Earth"}}}}});
218        assert_eq!(parse_entity(&body, "Q1").article.as_deref(), Some("Earth"));
219    }
220
221    #[test]
222    fn image_credit_is_plain_text() {
223        let body = json!({"query": {"pages": [{"imageinfo": [{
224            "thumburl": "https://upload.wikimedia.org/thumb.jpg",
225            "extmetadata": {
226                "Artist": {"value": "<a rel=\"nofollow\" href=\"https://flickr.com/x\">Jason  Mouratides</a>"},
227                "LicenseShortName": {"value": "CC BY 2.0"}
228            }
229        }]}]}});
230        assert_eq!(
231            parse_image(&body),
232            Some(Image {
233                url: "https://upload.wikimedia.org/thumb.jpg".into(),
234                credit: Some("Jason Mouratides · CC BY 2.0".into()),
235            })
236        );
237    }
238
239    #[test]
240    fn a_missing_file_is_no_image() {
241        let body = json!({"query": {"pages": [{"missing": true}]}});
242        assert_eq!(parse_image(&body), None);
243    }
244}