11#!/usr/bin/env python3
22"""Regenerate _data/publications.yml from DBLP, keeping manual edits intact.
33
4- Every AML Lab paper lists Rafet Sifa as a co-author, so his DBLP author page is
5- used as the lab's publication feed. Run with no arguments:
4+ Every AML Lab paper lists Rafet Sifa as a co-author, so his DBLP record is used
5+ as the lab's publication feed. Run with no arguments:
66
77 python3 scripts/fetch_publications.py
88
9+ Data comes from DBLP's SPARQL endpoint rather than the per-author XML export,
10+ because in September 2026 DBLP put its main site (the XML export and the search
11+ API included) behind an Anubis proof-of-work bot check, which no unattended
12+ script can pass. sparql.dblp.org serves the same knowledge graph unchallenged.
13+
914Hand-editing the generated file is supported in two ways:
1015
1116* ``manual: true`` freezes an entry. It is never overwritten, and it survives
3035import sys
3136import time
3237import urllib .error
38+ import urllib .parse
3339import urllib .request
34- import xml .etree .ElementTree as ET
3540from pathlib import Path
3641
3742try :
6873)
6974
7075OUTPUT = Path (__file__ ).resolve ().parent .parent / "_data" / "publications.yml"
71- DBLP_URL = f"https://dblp.org/pid/{ DBLP_PID } .xml"
76+ SPARQL_URL = "https://sparql.dblp.org/sparql"
77+ REC_PREFIX = "https://dblp.org/rec/"
7278USER_AGENT = "AMLLab-Publications-Bot/1.0 (+https://appliedmachinelearning-lab.github.io)"
7379
7480# DBLP appends a four-digit suffix to homonymous author names ("Kang Liu 0001").
7581HOMONYM_SUFFIX = re .compile (r"\s+\d{4}$" )
76- # Booktitles carry the proceedings volume for multi-volume conferences ("ECIR (3)").
77- # "(Findings)" and "(Industry)" are meaningful and must survive.
82+ # Some venue labels carry the proceedings volume for multi-volume conferences
83+ # ("ECIR (3)"). "(Findings)" and "(Industry)" are meaningful and must survive.
7884VOLUME_SUFFIX = re .compile (r"\s*\(\d+\)$" )
85+ # Separators for packing ordered author names into one GROUP_CONCAT string.
86+ # Chosen so they cannot occur inside a DBLP author name.
87+ AUTHOR_SEP , ORDINAL_SEP = "@@" , "~~"
88+
89+ QUERY = f"""
90+ PREFIX dblp: <https://dblp.org/rdf/schema#>
91+ PREFIX xsd: <http://www.w3.org/2001/XMLSchema#>
92+ SELECT ?pub ?year ?type
93+ (SAMPLE(?t) AS ?title) (SAMPLE(?bt) AS ?bibtex) (SAMPLE(?v) AS ?venue)
94+ (SAMPLE(?d) AS ?doi) (SAMPLE(?dp) AS ?docpage)
95+ (SAMPLE(?vol) AS ?volume) (SAMPLE(?toc) AS ?tocpage)
96+ (SAMPLE(?sch) AS ?school) (SAMPLE(?pby) AS ?publisher)
97+ (GROUP_CONCAT(CONCAT(STR(?ord), "{ ORDINAL_SEP } ", ?name);
98+ separator="{ AUTHOR_SEP } ") AS ?authors)
99+ WHERE {{
100+ ?pub dblp:authoredBy <https://dblp.org/pid/{ DBLP_PID } > ;
101+ dblp:title ?t ;
102+ dblp:yearOfPublication ?year ;
103+ a ?type .
104+ FILTER(?type != dblp:Publication)
105+ FILTER(xsd:integer(STR(?year)) >= { START_YEAR } )
106+ OPTIONAL {{ ?pub dblp:bibtexType ?bt }}
107+ OPTIONAL {{ ?pub dblp:publishedIn ?v }}
108+ OPTIONAL {{ ?pub dblp:doi ?d }}
109+ OPTIONAL {{ ?pub dblp:primaryDocumentPage ?dp }}
110+ OPTIONAL {{ ?pub dblp:publishedInJournalVolume ?vol }}
111+ OPTIONAL {{ ?pub dblp:listedOnTocPage ?toc }}
112+ OPTIONAL {{ ?pub dblp:thesisAcceptedBySchool ?sch }}
113+ OPTIONAL {{ ?pub dblp:publishedBy ?pby }}
114+ ?pub dblp:hasSignature ?sig .
115+ ?sig dblp:signatureOrdinal ?ord ;
116+ dblp:signatureDblpName ?name .
117+ }}
118+ GROUP BY ?pub ?year ?type
119+ """
79120
80121
81- def fetch (url : str , attempts : int = 4 ) -> bytes :
82- """GET url, backing off on the 429s DBLP hands out to impatient clients."""
83- request = urllib .request .Request (url , headers = {"User-Agent" : USER_AGENT })
122+ def fetch (attempts : int = 4 ) -> dict :
123+ """Run the SPARQL query, backing off on the endpoint's transient errors."""
124+ body = urllib .parse .urlencode ({"query" : QUERY }).encode ()
125+ request = urllib .request .Request (
126+ SPARQL_URL ,
127+ data = body ,
128+ headers = {
129+ "User-Agent" : USER_AGENT ,
130+ "Accept" : "application/sparql-results+json" ,
131+ "Content-Type" : "application/x-www-form-urlencoded" ,
132+ },
133+ )
84134 for attempt in range (1 , attempts + 1 ):
85135 try :
86- with urllib .request .urlopen (request , timeout = 60 ) as response :
87- return response .read ()
88- except (urllib .error .URLError , TimeoutError ) as exc :
89- retryable = getattr (exc , "code" , None ) in (429 , 500 , 502 , 503 , 504 )
90- if attempt == attempts or not (retryable or isinstance (exc , (urllib .error .URLError , TimeoutError ))):
136+ with urllib .request .urlopen (request , timeout = 120 ) as response :
137+ raw = response .read ()
138+ try :
139+ return json .loads (raw )
140+ except json .JSONDecodeError :
141+ # A bot-check or maintenance page arrives as HTML with status 200,
142+ # so a bad body has to be treated as a failure in its own right.
143+ snippet = raw [:200 ].decode ("utf-8" , "replace" ).replace ("\n " , " " )
144+ raise RuntimeError (f"expected JSON, got: { snippet } " )
145+ except (urllib .error .URLError , TimeoutError , RuntimeError ) as exc :
146+ if attempt == attempts :
91147 raise
92148 delay = 5 * 2 ** (attempt - 1 )
93149 print (f" { exc } — retrying in { delay } s" , file = sys .stderr )
94150 time .sleep (delay )
95151 raise RuntimeError ("unreachable" )
96152
97153
98- def text_of (element : ET .Element | None ) -> str :
99- """Flatten an element's text, dropping the inline <i>/<sub> markup DBLP uses."""
100- if element is None :
101- return ""
102- return re .sub (r"\s+" , " " , "" .join (element .itertext ())).strip ()
103-
104-
105154def clean_title (raw : str ) -> str :
106- return raw .rstrip ("." ).strip () if raw .endswith ("." ) else raw
155+ raw = re .sub (r"\s+" , " " , raw ).strip ()
156+ return raw [:- 1 ].strip () if raw .endswith ("." ) else raw
107157
108158
109159def normalize_title (raw : str ) -> str :
110160 """Key for matching a preprint against its published version."""
111161 return re .sub (r"[^a-z0-9]" , "" , raw .lower ())
112162
113163
114- def dblp_link (entry : ET .Element ) -> str :
115- """DBLP's <url> is relative to the site root ("db/conf/...#key")."""
116- url = text_of (entry .find ("url" ))
117- return f"https://dblp.org/{ url } " if url else ""
118-
119-
120- def best_link (entry : ET .Element ) -> str :
121- """Prefer a DOI, then any other publisher link, then the DBLP record."""
122- links = [text_of (ee ) for ee in entry .findall ("ee" ) if text_of (ee )]
123- for link in links :
124- if "doi.org" in link :
125- return link
126- if links :
127- return links [0 ]
128- return dblp_link (entry )
129-
130-
131- def venue_of (entry : ET .Element ) -> str :
132- if entry .tag == "inproceedings" :
133- return VOLUME_SUFFIX .sub ("" , text_of (entry .find ("booktitle" )))
134- if entry .tag == "article" :
135- journal = text_of (entry .find ("journal" ))
136- # The template already tags these as preprints, so "arXiv" alone reads better.
137- return "arXiv" if journal == "CoRR" else journal
138- if entry .tag == "book" :
139- parts = [text_of (entry .find ("series" )), text_of (entry .find ("publisher" ))]
140- return ", " .join (part for part in parts if part )
141- if entry .tag == "phdthesis" :
142- # Theses are single-authored, so they only reach DBLP's feed for Rafet's
143- # own; lab members' theses have to be added by hand with manual: true.
144- return text_of (entry .find ("school" ))
145- return ""
146-
147-
148- def parse (xml : bytes ) -> list [dict ]:
149- root = ET .fromstring (xml )
164+ def authors_of (packed : str ) -> list [str ]:
165+ """Unpack the "<ordinal>~~<name>@@..." blob into author order."""
166+ people = []
167+ for item in packed .split (AUTHOR_SEP ):
168+ ordinal , _ , name = item .partition (ORDINAL_SEP )
169+ if not name :
170+ continue
171+ try :
172+ rank = int (ordinal )
173+ except ValueError :
174+ rank = 999
175+ people .append ((rank , HOMONYM_SUFFIX .sub ("" , name .strip ())))
176+ return [name for _ , name in sorted (people )]
177+
178+
179+ def parse (payload : dict ) -> list [dict ]:
150180 peer_reviewed : list [dict ] = []
151181 preprints : list [dict ] = []
152182
153- for record in root .findall ("r" ):
154- for entry in record :
155- year = entry .findtext ("year" )
156- if not year or int (year ) < START_YEAR :
157- continue
158-
159- title = clean_title (text_of (entry .find ("title" )))
160- if not title :
161- continue
162-
163- is_preprint = entry .get ("publtype" ) == "informal"
164- arxiv_id = ""
165- if is_preprint :
166- # CoRR entries store the arXiv id in <volume> as "abs/2601.14039".
167- arxiv_id = text_of (entry .find ("volume" )).removeprefix ("abs/" )
168-
169- publication = {
170- "key" : entry .get ("key" , "" ),
171- "title" : title ,
172- "authors" : [
173- HOMONYM_SUFFIX .sub ("" , text_of (author ))
174- for author in entry .findall ("author" )
175- ],
176- "year" : int (year ),
177- "venue" : venue_of (entry ),
178- "type" : "preprint" if is_preprint else entry .tag ,
179- "url" : f"https://arxiv.org/abs/{ arxiv_id } " if arxiv_id else best_link (entry ),
180- "dblp" : dblp_link (entry ),
181- }
182- (preprints if is_preprint else peer_reviewed ).append (publication )
183+ for row in payload ["results" ]["bindings" ]:
184+ def value (name : str ) -> str :
185+ return row .get (name , {}).get ("value" , "" ).strip ()
186+
187+ title = clean_title (value ("title" ))
188+ key = value ("pub" ).removeprefix (REC_PREFIX )
189+ if not title or not key :
190+ continue
191+
192+ rdf_type = value ("type" ).rsplit ("#" , 1 )[- 1 ]
193+ is_thesis = value ("bibtex" ).endswith ("Phdthesis" )
194+ is_preprint = rdf_type == "Informal"
195+ kind = {
196+ "Inproceedings" : "inproceedings" ,
197+ "Article" : "article" ,
198+ "Informal" : "preprint" ,
199+ "Book" : "phdthesis" if is_thesis else "book" ,
200+ }.get (rdf_type , "inproceedings" )
201+
202+ # CoRR records carry the arXiv id in the journal volume ("abs/2601.14039").
203+ arxiv_id = value ("volume" ).removeprefix ("abs/" ) if is_preprint else ""
204+
205+ if is_preprint :
206+ venue = "arXiv"
207+ elif is_thesis :
208+ venue = value ("school" )
209+ elif kind == "book" :
210+ # Books read better with their publisher: "Cognitive Technologies, Springer".
211+ venue = ", " .join (p for p in (value ("venue" ), value ("publisher" )) if p )
212+ else :
213+ venue = VOLUME_SUFFIX .sub ("" , value ("venue" ))
214+
215+ # DBLP's TOC page plus the record key is the human-facing record link.
216+ toc = value ("tocpage" )
217+ dblp_link = f"{ toc } .html#{ key .rsplit ('/' , 1 )[- 1 ]} " if toc else ""
218+
219+ if arxiv_id :
220+ url = f"https://arxiv.org/abs/{ arxiv_id } "
221+ else :
222+ url = value ("doi" ) or value ("docpage" ) or dblp_link
223+
224+ publication = {
225+ "key" : key ,
226+ "title" : title ,
227+ "authors" : authors_of (value ("authors" )),
228+ "year" : int (value ("year" )),
229+ "venue" : venue ,
230+ "type" : kind ,
231+ "url" : url ,
232+ "dblp" : dblp_link ,
233+ }
234+ (preprints if is_preprint else peer_reviewed ).append (publication )
183235
184236 # A preprint that later appeared at a venue would otherwise be listed twice.
185237 published_titles = {normalize_title (p ["title" ]) for p in peer_reviewed }
@@ -304,7 +356,7 @@ def to_yaml(publications: list[dict]) -> str:
304356 """Emit YAML by hand so the field order and quoting stay diff-friendly."""
305357 lines = [
306358 "# Generated by scripts/fetch_publications.py from DBLP — refreshed weekly." ,
307- f"# Source: https:// dblp.org/ pid/ { DBLP_PID } .html (papers from { START_YEAR } onwards)" ,
359+ f"# Source: sparql. dblp.org, author pid { DBLP_PID } (papers from { START_YEAR } onwards)" ,
308360 "#" ,
309361 "# Hand edits: set `manual: true` on an entry to freeze it (never overwritten," ,
310362 "# and kept even if DBLP drops it). `oa_url:` is preserved on every refresh" ,
@@ -330,8 +382,8 @@ def to_yaml(publications: list[dict]) -> str:
330382
331383
332384def main () -> int :
333- print (f"Fetching { DBLP_URL } " )
334- publications = merge (parse (fetch (DBLP_URL )), load_existing (OUTPUT ))
385+ print (f"Querying { SPARQL_URL } for pid { DBLP_PID } " )
386+ publications = merge (parse (fetch ()), load_existing (OUTPUT ))
335387 if not publications :
336388 print ("Refusing to write an empty publication list." , file = sys .stderr )
337389 return 1
0 commit comments