inspire

InspireHEP REST API v1 client.

bib_checker.inspire

InspireHEP REST API v1 client.

Notes

API documentation: https://inspirehep.net/api

InspireClient

Thin wrapper around the InspireHEP literature API.

Source code in src/bib_checker/inspire.py
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
class InspireClient:
    """Thin wrapper around the InspireHEP literature API."""

    def __init__(
        self,
        timeout: float = _DEFAULT_TIMEOUT,
        rate_limit_delay: float = _RATE_LIMIT_DELAY,
    ) -> None:
        """Create a new client.

        Parameters
        ----------
        timeout : float, optional
            Per-request timeout in seconds. Default is 15.
        rate_limit_delay : float, optional
            Seconds to wait between consecutive requests. Default is 0.5.
        """
        self._timeout = timeout
        self._delay = rate_limit_delay
        self._session = requests.Session()
        self._session.headers["User-Agent"] = (
            "bib-checker/0.1 (https://github.com/MeighenBergerS/bib-checker)"
        )

    # ------------------------------------------------------------------
    # Public interface
    # ------------------------------------------------------------------

    def lookup_by_texkey(self, texkey: str) -> dict[str, Any] | None:
        """Return the first InspireHEP record matching *texkey*, or None.

        Parameters
        ----------
        texkey : str
            InspireHEP texkey, e.g. ``Spolyar:2007qv``.

        Returns
        -------
        record : dict[str, Any] or None
            Raw API hit dict, or ``None`` if no record was found.
        """
        results = self.lookup_by_texkeys([texkey])
        return results.get(texkey)

    def lookup_by_texkeys(
        self,
        texkeys: list[str],
        batch_size: int = 50,
    ) -> dict[str, dict[str, Any]]:
        """Fetch multiple records in batches using OR queries.

        Parameters
        ----------
        texkeys : list[str]
            Citation keys to look up.
        batch_size : int, optional
            Maximum number of keys per API request. Default is 50.

        Returns
        -------
        records : dict[str, dict[str, Any]]
            Mapping of texkey to the matching raw API hit dict.
            Keys not found on InspireHEP are absent from the result.
        """
        records: dict[str, dict[str, Any]] = {}
        for i in range(0, len(texkeys), batch_size):
            chunk = texkeys[i : i + batch_size]
            query = " or ".join(f"texkey:{k}" for k in chunk)
            params = {
                "q": query,
                "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info,imprint",
                "size": batch_size,
            }
            data = self._get(f"{_BASE}/literature", params=params)
            for hit in data.get("hits", {}).get("hits", []):
                key = self.get_texkey(hit)
                if key:
                    records[key] = hit
        return records

    def lookup_by_ads_bibcode(self, bibcode: str) -> dict[str, Any] | None:
        """Return the first InspireHEP record matching an ADS bibcode, or None.

        Parameters
        ----------
        bibcode : str
            ADS bibcode extracted from an ``adsurl`` field, e.g.
            ``2019ApJS..243...10P``.

        Returns
        -------
        record : dict[str, Any] or None
            Raw API hit dict, or ``None`` if no record was found.
        """
        params = {
            "q": f"external_system_identifiers.value:{bibcode}",
            "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info,imprint",
            "size": 1,
        }
        data = self._get(f"{_BASE}/literature", params=params)
        hits = data.get("hits", {}).get("hits", [])
        return hits[0] if hits else None

    def search(
        self,
        query: str,
        size: int = 5,
    ) -> list[dict[str, Any]]:
        """Search InspireHEP and return up to *size* records.

        Parameters
        ----------
        query : str
            Free-text InspireHEP search query.
        size : int, optional
            Maximum number of results to return. Default is 5.

        Returns
        -------
        hits : list[dict[str, Any]]
            Raw API hit dicts ordered by most-recent.
        """
        params = {
            "q": query,
            "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info",
            "size": size,
            "sort": "mostrecent",
        }
        data = self._get(f"{_BASE}/literature", params=params)
        return data.get("hits", {}).get("hits", [])

    # ------------------------------------------------------------------
    # Helpers to extract normalised values from raw API records
    # ------------------------------------------------------------------

    @staticmethod
    def get_texkey(record: dict[str, Any]) -> str:
        """Extract the primary texkey from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        texkey : str
            Primary texkey, or an empty string if absent.
        """
        keys = record.get("metadata", {}).get("texkeys", [])
        return keys[0] if keys else ""

    @staticmethod
    def get_title(record: dict[str, Any]) -> str:
        """Extract the primary title from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        title : str
            Primary title string, or an empty string if absent.
        """
        titles = record.get("metadata", {}).get("titles", [])
        return titles[0].get("title", "") if titles else ""

    @staticmethod
    def get_doi(record: dict[str, Any]) -> str:
        """Extract the primary DOI from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        doi : str
            Primary DOI string, or an empty string if absent.
        """
        dois = record.get("metadata", {}).get("dois", [])
        return dois[0].get("value", "") if dois else ""

    @staticmethod
    def get_eprint(record: dict[str, Any]) -> str:
        """Extract the primary ArXiv eprint ID from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        eprint : str
            ArXiv eprint identifier, or an empty string if absent.
        """
        eprints = record.get("metadata", {}).get("arxiv_eprints", [])
        return eprints[0].get("value", "") if eprints else ""

    @staticmethod
    def get_year(record: dict[str, Any]) -> str:
        """Extract the publication year from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        year : str
            Four-digit year string, or an empty string if absent.
        """
        pub = record.get("metadata", {}).get("publication_info", [])
        if pub:
            return str(pub[0].get("year", ""))
        imprint = record.get("metadata", {}).get("imprint", [])
        if imprint:
            return str(imprint[0].get("date", ""))[:4]
        return ""

    @staticmethod
    def get_authors(record: dict[str, Any]) -> list[str]:
        """Extract author full names from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        authors : list[str]
            Full name strings for each author.
        """
        authors = record.get("metadata", {}).get("authors", [])
        return [a.get("full_name", "") for a in authors]

    @staticmethod
    def get_inspire_id(record: dict[str, Any]) -> str:
        """Extract the InspireHEP internal record ID from a raw API record.

        Parameters
        ----------
        record : dict[str, Any]
            Raw API hit dict.

        Returns
        -------
        inspire_id : str
            Record ID as a string, or an empty string if absent.
        """
        return str(record.get("id", ""))

    # ------------------------------------------------------------------
    # Internal
    # ------------------------------------------------------------------

    def _get(self, url: str, params: dict[str, Any]) -> dict[str, Any]:
        """Send a GET request and return the parsed JSON response.

        Parameters
        ----------
        url : str
            Request URL.
        params : dict[str, Any]
            Query parameters.

        Returns
        -------
        data : dict[str, Any]
            Parsed JSON response body.
        """
        time.sleep(self._delay)
        response = self._session.get(url, params=params, timeout=self._timeout)
        response.raise_for_status()
        return response.json()

__init__

__init__(
    timeout=_DEFAULT_TIMEOUT,
    rate_limit_delay=_RATE_LIMIT_DELAY,
)

Create a new client.

Parameters:
  • timeout (float, default: _DEFAULT_TIMEOUT ) –

    Per-request timeout in seconds. Default is 15.

  • rate_limit_delay (float, default: _RATE_LIMIT_DELAY ) –

    Seconds to wait between consecutive requests. Default is 0.5.

Source code in src/bib_checker/inspire.py
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
def __init__(
    self,
    timeout: float = _DEFAULT_TIMEOUT,
    rate_limit_delay: float = _RATE_LIMIT_DELAY,
) -> None:
    """Create a new client.

    Parameters
    ----------
    timeout : float, optional
        Per-request timeout in seconds. Default is 15.
    rate_limit_delay : float, optional
        Seconds to wait between consecutive requests. Default is 0.5.
    """
    self._timeout = timeout
    self._delay = rate_limit_delay
    self._session = requests.Session()
    self._session.headers["User-Agent"] = (
        "bib-checker/0.1 (https://github.com/MeighenBergerS/bib-checker)"
    )

get_authors staticmethod

get_authors(record)

Extract author full names from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • authors( list[str] ) –

    Full name strings for each author.

Source code in src/bib_checker/inspire.py
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
@staticmethod
def get_authors(record: dict[str, Any]) -> list[str]:
    """Extract author full names from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    authors : list[str]
        Full name strings for each author.
    """
    authors = record.get("metadata", {}).get("authors", [])
    return [a.get("full_name", "") for a in authors]

get_doi staticmethod

get_doi(record)

Extract the primary DOI from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • doi( str ) –

    Primary DOI string, or an empty string if absent.

Source code in src/bib_checker/inspire.py
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
@staticmethod
def get_doi(record: dict[str, Any]) -> str:
    """Extract the primary DOI from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    doi : str
        Primary DOI string, or an empty string if absent.
    """
    dois = record.get("metadata", {}).get("dois", [])
    return dois[0].get("value", "") if dois else ""

get_eprint staticmethod

get_eprint(record)

Extract the primary ArXiv eprint ID from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • eprint( str ) –

    ArXiv eprint identifier, or an empty string if absent.

Source code in src/bib_checker/inspire.py
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
@staticmethod
def get_eprint(record: dict[str, Any]) -> str:
    """Extract the primary ArXiv eprint ID from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    eprint : str
        ArXiv eprint identifier, or an empty string if absent.
    """
    eprints = record.get("metadata", {}).get("arxiv_eprints", [])
    return eprints[0].get("value", "") if eprints else ""

get_inspire_id staticmethod

get_inspire_id(record)

Extract the InspireHEP internal record ID from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • inspire_id( str ) –

    Record ID as a string, or an empty string if absent.

Source code in src/bib_checker/inspire.py
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
@staticmethod
def get_inspire_id(record: dict[str, Any]) -> str:
    """Extract the InspireHEP internal record ID from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    inspire_id : str
        Record ID as a string, or an empty string if absent.
    """
    return str(record.get("id", ""))

get_texkey staticmethod

get_texkey(record)

Extract the primary texkey from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • texkey( str ) –

    Primary texkey, or an empty string if absent.

Source code in src/bib_checker/inspire.py
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
@staticmethod
def get_texkey(record: dict[str, Any]) -> str:
    """Extract the primary texkey from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    texkey : str
        Primary texkey, or an empty string if absent.
    """
    keys = record.get("metadata", {}).get("texkeys", [])
    return keys[0] if keys else ""

get_title staticmethod

get_title(record)

Extract the primary title from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • title( str ) –

    Primary title string, or an empty string if absent.

Source code in src/bib_checker/inspire.py
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
@staticmethod
def get_title(record: dict[str, Any]) -> str:
    """Extract the primary title from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    title : str
        Primary title string, or an empty string if absent.
    """
    titles = record.get("metadata", {}).get("titles", [])
    return titles[0].get("title", "") if titles else ""

get_year staticmethod

get_year(record)

Extract the publication year from a raw API record.

Parameters:
  • record (dict[str, Any]) –

    Raw API hit dict.

Returns:
  • year( str ) –

    Four-digit year string, or an empty string if absent.

Source code in src/bib_checker/inspire.py
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
@staticmethod
def get_year(record: dict[str, Any]) -> str:
    """Extract the publication year from a raw API record.

    Parameters
    ----------
    record : dict[str, Any]
        Raw API hit dict.

    Returns
    -------
    year : str
        Four-digit year string, or an empty string if absent.
    """
    pub = record.get("metadata", {}).get("publication_info", [])
    if pub:
        return str(pub[0].get("year", ""))
    imprint = record.get("metadata", {}).get("imprint", [])
    if imprint:
        return str(imprint[0].get("date", ""))[:4]
    return ""

lookup_by_ads_bibcode

lookup_by_ads_bibcode(bibcode)

Return the first InspireHEP record matching an ADS bibcode, or None.

Parameters:
  • bibcode (str) –

    ADS bibcode extracted from an adsurl field, e.g. 2019ApJS..243...10P.

Returns:
  • record( dict[str, Any] or None ) –

    Raw API hit dict, or None if no record was found.

Source code in src/bib_checker/inspire.py
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
def lookup_by_ads_bibcode(self, bibcode: str) -> dict[str, Any] | None:
    """Return the first InspireHEP record matching an ADS bibcode, or None.

    Parameters
    ----------
    bibcode : str
        ADS bibcode extracted from an ``adsurl`` field, e.g.
        ``2019ApJS..243...10P``.

    Returns
    -------
    record : dict[str, Any] or None
        Raw API hit dict, or ``None`` if no record was found.
    """
    params = {
        "q": f"external_system_identifiers.value:{bibcode}",
        "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info,imprint",
        "size": 1,
    }
    data = self._get(f"{_BASE}/literature", params=params)
    hits = data.get("hits", {}).get("hits", [])
    return hits[0] if hits else None

lookup_by_texkey

lookup_by_texkey(texkey)

Return the first InspireHEP record matching texkey, or None.

Parameters:
  • texkey (str) –

    InspireHEP texkey, e.g. Spolyar:2007qv.

Returns:
  • record( dict[str, Any] or None ) –

    Raw API hit dict, or None if no record was found.

Source code in src/bib_checker/inspire.py
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
def lookup_by_texkey(self, texkey: str) -> dict[str, Any] | None:
    """Return the first InspireHEP record matching *texkey*, or None.

    Parameters
    ----------
    texkey : str
        InspireHEP texkey, e.g. ``Spolyar:2007qv``.

    Returns
    -------
    record : dict[str, Any] or None
        Raw API hit dict, or ``None`` if no record was found.
    """
    results = self.lookup_by_texkeys([texkey])
    return results.get(texkey)

lookup_by_texkeys

lookup_by_texkeys(texkeys, batch_size=50)

Fetch multiple records in batches using OR queries.

Parameters:
  • texkeys (list[str]) –

    Citation keys to look up.

  • batch_size (int, default: 50 ) –

    Maximum number of keys per API request. Default is 50.

Returns:
  • records( dict[str, dict[str, Any]] ) –

    Mapping of texkey to the matching raw API hit dict. Keys not found on InspireHEP are absent from the result.

Source code in src/bib_checker/inspire.py
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
def lookup_by_texkeys(
    self,
    texkeys: list[str],
    batch_size: int = 50,
) -> dict[str, dict[str, Any]]:
    """Fetch multiple records in batches using OR queries.

    Parameters
    ----------
    texkeys : list[str]
        Citation keys to look up.
    batch_size : int, optional
        Maximum number of keys per API request. Default is 50.

    Returns
    -------
    records : dict[str, dict[str, Any]]
        Mapping of texkey to the matching raw API hit dict.
        Keys not found on InspireHEP are absent from the result.
    """
    records: dict[str, dict[str, Any]] = {}
    for i in range(0, len(texkeys), batch_size):
        chunk = texkeys[i : i + batch_size]
        query = " or ".join(f"texkey:{k}" for k in chunk)
        params = {
            "q": query,
            "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info,imprint",
            "size": batch_size,
        }
        data = self._get(f"{_BASE}/literature", params=params)
        for hit in data.get("hits", {}).get("hits", []):
            key = self.get_texkey(hit)
            if key:
                records[key] = hit
    return records

search

search(query, size=5)

Search InspireHEP and return up to size records.

Parameters:
  • query (str) –

    Free-text InspireHEP search query.

  • size (int, default: 5 ) –

    Maximum number of results to return. Default is 5.

Returns:
  • hits( list[dict[str, Any]] ) –

    Raw API hit dicts ordered by most-recent.

Source code in src/bib_checker/inspire.py
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
def search(
    self,
    query: str,
    size: int = 5,
) -> list[dict[str, Any]]:
    """Search InspireHEP and return up to *size* records.

    Parameters
    ----------
    query : str
        Free-text InspireHEP search query.
    size : int, optional
        Maximum number of results to return. Default is 5.

    Returns
    -------
    hits : list[dict[str, Any]]
        Raw API hit dicts ordered by most-recent.
    """
    params = {
        "q": query,
        "fields": "texkeys,titles,authors,dois,arxiv_eprints,publication_info",
        "size": size,
        "sort": "mostrecent",
    }
    data = self._get(f"{_BASE}/literature", params=params)
    return data.get("hits", {}).get("hits", [])