Skip to content

Commit 6254459

Browse files
committed
Merge branch 'dblp_publications_fix_via_sparql'
2 parents cc6d70e + a0d77bf commit 6254459

2 files changed

Lines changed: 144 additions & 92 deletions

File tree

‎README.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -44,7 +44,7 @@ The post will automatically appear on the [/news/](/news/) page. The date in the
4444

4545
## Publications
4646

47-
The list on [/research/publications](/research/publications) is generated from Rafet's DBLP author page, which works as the lab's feed because every lab paper lists him as a co-author. A GitHub Action re-runs `scripts/fetch_publications.py` every Monday and commits `_data/publications.yml` only if something changed, so new papers show up on their own within a week of DBLP indexing them. To pull them in sooner, go to the **Actions** tab -> **Update publications** -> **Run workflow**.
47+
The list on [/research/publications](/research/publications) is generated from Rafet's DBLP record, which works as the lab's feed because every lab paper lists him as a co-author. The data is read from DBLP's SPARQL endpoint (`sparql.dblp.org`), because DBLP put its main site, the per-author XML export and the search API included, behind a proof-of-work bot check in September 2026 (we can reevaluate this later). A GitHub Action re-runs `scripts/fetch_publications.py` every Monday and commits `_data/publications.yml` only if something changed, so new papers show up on their own within a week of DBLP indexing them. To pull them in sooner, go to the **Actions** tab -> **Update publications** -> **Run workflow**.
4848

4949
Everything below is done by editing `_data/publications.yml` and committing it to `main`. Your edits are kept on the next refresh, the script merges them back in rather than overwriting the file blindly. Note that YAML comments in that file are *not* preserved, since it gets rewritten on every run.
5050

‎scripts/fetch_publications.py‎

Lines changed: 143 additions & 91 deletions
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,16 @@
11
#!/usr/bin/env python3
22
"""Regenerate _data/publications.yml from DBLP, keeping manual edits intact.
33
4-
Every AML Lab paper lists Rafet Sifa as a co-author, so his DBLP author page is
5-
used as the lab's publication feed. Run with no arguments:
4+
Every AML Lab paper lists Rafet Sifa as a co-author, so his DBLP record is used
5+
as the lab's publication feed. Run with no arguments:
66
77
python3 scripts/fetch_publications.py
88
9+
Data comes from DBLP's SPARQL endpoint rather than the per-author XML export,
10+
because in September 2026 DBLP put its main site (the XML export and the search
11+
API included) behind an Anubis proof-of-work bot check, which no unattended
12+
script can pass. sparql.dblp.org serves the same knowledge graph unchallenged.
13+
914
Hand-editing the generated file is supported in two ways:
1015
1116
* ``manual: true`` freezes an entry. It is never overwritten, and it survives
@@ -30,8 +35,8 @@
3035
import sys
3136
import time
3237
import urllib.error
38+
import urllib.parse
3339
import urllib.request
34-
import xml.etree.ElementTree as ET
3540
from pathlib import Path
3641

3742
try:
@@ -68,118 +73,165 @@
6873
)
6974

7075
OUTPUT = Path(__file__).resolve().parent.parent / "_data" / "publications.yml"
71-
DBLP_URL = f"https://dblp.org/pid/{DBLP_PID}.xml"
76+
SPARQL_URL = "https://sparql.dblp.org/sparql"
77+
REC_PREFIX = "https://dblp.org/rec/"
7278
USER_AGENT = "AMLLab-Publications-Bot/1.0 (+https://appliedmachinelearning-lab.github.io)"
7379

7480
# DBLP appends a four-digit suffix to homonymous author names ("Kang Liu 0001").
7581
HOMONYM_SUFFIX = re.compile(r"\s+\d{4}$")
76-
# Booktitles carry the proceedings volume for multi-volume conferences ("ECIR (3)").
77-
# "(Findings)" and "(Industry)" are meaningful and must survive.
82+
# Some venue labels carry the proceedings volume for multi-volume conferences
83+
# ("ECIR (3)"). "(Findings)" and "(Industry)" are meaningful and must survive.
7884
VOLUME_SUFFIX = re.compile(r"\s*\(\d+\)$")
85+
# Separators for packing ordered author names into one GROUP_CONCAT string.
86+
# Chosen so they cannot occur inside a DBLP author name.
87+
AUTHOR_SEP, ORDINAL_SEP = "@@", "~~"
88+
89+
QUERY = f"""
90+
PREFIX dblp: <https://dblp.org/rdf/schema#>
91+
PREFIX xsd: <http://www.w3.org/2001/XMLSchema#>
92+
SELECT ?pub ?year ?type
93+
(SAMPLE(?t) AS ?title) (SAMPLE(?bt) AS ?bibtex) (SAMPLE(?v) AS ?venue)
94+
(SAMPLE(?d) AS ?doi) (SAMPLE(?dp) AS ?docpage)
95+
(SAMPLE(?vol) AS ?volume) (SAMPLE(?toc) AS ?tocpage)
96+
(SAMPLE(?sch) AS ?school) (SAMPLE(?pby) AS ?publisher)
97+
(GROUP_CONCAT(CONCAT(STR(?ord), "{ORDINAL_SEP}", ?name);
98+
separator="{AUTHOR_SEP}") AS ?authors)
99+
WHERE {{
100+
?pub dblp:authoredBy <https://dblp.org/pid/{DBLP_PID}> ;
101+
dblp:title ?t ;
102+
dblp:yearOfPublication ?year ;
103+
a ?type .
104+
FILTER(?type != dblp:Publication)
105+
FILTER(xsd:integer(STR(?year)) >= {START_YEAR})
106+
OPTIONAL {{ ?pub dblp:bibtexType ?bt }}
107+
OPTIONAL {{ ?pub dblp:publishedIn ?v }}
108+
OPTIONAL {{ ?pub dblp:doi ?d }}
109+
OPTIONAL {{ ?pub dblp:primaryDocumentPage ?dp }}
110+
OPTIONAL {{ ?pub dblp:publishedInJournalVolume ?vol }}
111+
OPTIONAL {{ ?pub dblp:listedOnTocPage ?toc }}
112+
OPTIONAL {{ ?pub dblp:thesisAcceptedBySchool ?sch }}
113+
OPTIONAL {{ ?pub dblp:publishedBy ?pby }}
114+
?pub dblp:hasSignature ?sig .
115+
?sig dblp:signatureOrdinal ?ord ;
116+
dblp:signatureDblpName ?name .
117+
}}
118+
GROUP BY ?pub ?year ?type
119+
"""
79120

80121

81-
def fetch(url: str, attempts: int = 4) -> bytes:
82-
"""GET url, backing off on the 429s DBLP hands out to impatient clients."""
83-
request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
122+
def fetch(attempts: int = 4) -> dict:
123+
"""Run the SPARQL query, backing off on the endpoint's transient errors."""
124+
body = urllib.parse.urlencode({"query": QUERY}).encode()
125+
request = urllib.request.Request(
126+
SPARQL_URL,
127+
data=body,
128+
headers={
129+
"User-Agent": USER_AGENT,
130+
"Accept": "application/sparql-results+json",
131+
"Content-Type": "application/x-www-form-urlencoded",
132+
},
133+
)
84134
for attempt in range(1, attempts + 1):
85135
try:
86-
with urllib.request.urlopen(request, timeout=60) as response:
87-
return response.read()
88-
except (urllib.error.URLError, TimeoutError) as exc:
89-
retryable = getattr(exc, "code", None) in (429, 500, 502, 503, 504)
90-
if attempt == attempts or not (retryable or isinstance(exc, (urllib.error.URLError, TimeoutError))):
136+
with urllib.request.urlopen(request, timeout=120) as response:
137+
raw = response.read()
138+
try:
139+
return json.loads(raw)
140+
except json.JSONDecodeError:
141+
# A bot-check or maintenance page arrives as HTML with status 200,
142+
# so a bad body has to be treated as a failure in its own right.
143+
snippet = raw[:200].decode("utf-8", "replace").replace("\n", " ")
144+
raise RuntimeError(f"expected JSON, got: {snippet}")
145+
except (urllib.error.URLError, TimeoutError, RuntimeError) as exc:
146+
if attempt == attempts:
91147
raise
92148
delay = 5 * 2 ** (attempt - 1)
93149
print(f" {exc} — retrying in {delay}s", file=sys.stderr)
94150
time.sleep(delay)
95151
raise RuntimeError("unreachable")
96152

97153

98-
def text_of(element: ET.Element | None) -> str:
99-
"""Flatten an element's text, dropping the inline <i>/<sub> markup DBLP uses."""
100-
if element is None:
101-
return ""
102-
return re.sub(r"\s+", " ", "".join(element.itertext())).strip()
103-
104-
105154
def clean_title(raw: str) -> str:
106-
return raw.rstrip(".").strip() if raw.endswith(".") else raw
155+
raw = re.sub(r"\s+", " ", raw).strip()
156+
return raw[:-1].strip() if raw.endswith(".") else raw
107157

108158

109159
def normalize_title(raw: str) -> str:
110160
"""Key for matching a preprint against its published version."""
111161
return re.sub(r"[^a-z0-9]", "", raw.lower())
112162

113163

114-
def dblp_link(entry: ET.Element) -> str:
115-
"""DBLP's <url> is relative to the site root ("db/conf/...#key")."""
116-
url = text_of(entry.find("url"))
117-
return f"https://dblp.org/{url}" if url else ""
118-
119-
120-
def best_link(entry: ET.Element) -> str:
121-
"""Prefer a DOI, then any other publisher link, then the DBLP record."""
122-
links = [text_of(ee) for ee in entry.findall("ee") if text_of(ee)]
123-
for link in links:
124-
if "doi.org" in link:
125-
return link
126-
if links:
127-
return links[0]
128-
return dblp_link(entry)
129-
130-
131-
def venue_of(entry: ET.Element) -> str:
132-
if entry.tag == "inproceedings":
133-
return VOLUME_SUFFIX.sub("", text_of(entry.find("booktitle")))
134-
if entry.tag == "article":
135-
journal = text_of(entry.find("journal"))
136-
# The template already tags these as preprints, so "arXiv" alone reads better.
137-
return "arXiv" if journal == "CoRR" else journal
138-
if entry.tag == "book":
139-
parts = [text_of(entry.find("series")), text_of(entry.find("publisher"))]
140-
return ", ".join(part for part in parts if part)
141-
if entry.tag == "phdthesis":
142-
# Theses are single-authored, so they only reach DBLP's feed for Rafet's
143-
# own; lab members' theses have to be added by hand with manual: true.
144-
return text_of(entry.find("school"))
145-
return ""
146-
147-
148-
def parse(xml: bytes) -> list[dict]:
149-
root = ET.fromstring(xml)
164+
def authors_of(packed: str) -> list[str]:
165+
"""Unpack the "<ordinal>~~<name>@@..." blob into author order."""
166+
people = []
167+
for item in packed.split(AUTHOR_SEP):
168+
ordinal, _, name = item.partition(ORDINAL_SEP)
169+
if not name:
170+
continue
171+
try:
172+
rank = int(ordinal)
173+
except ValueError:
174+
rank = 999
175+
people.append((rank, HOMONYM_SUFFIX.sub("", name.strip())))
176+
return [name for _, name in sorted(people)]
177+
178+
179+
def parse(payload: dict) -> list[dict]:
150180
peer_reviewed: list[dict] = []
151181
preprints: list[dict] = []
152182

153-
for record in root.findall("r"):
154-
for entry in record:
155-
year = entry.findtext("year")
156-
if not year or int(year) < START_YEAR:
157-
continue
158-
159-
title = clean_title(text_of(entry.find("title")))
160-
if not title:
161-
continue
162-
163-
is_preprint = entry.get("publtype") == "informal"
164-
arxiv_id = ""
165-
if is_preprint:
166-
# CoRR entries store the arXiv id in <volume> as "abs/2601.14039".
167-
arxiv_id = text_of(entry.find("volume")).removeprefix("abs/")
168-
169-
publication = {
170-
"key": entry.get("key", ""),
171-
"title": title,
172-
"authors": [
173-
HOMONYM_SUFFIX.sub("", text_of(author))
174-
for author in entry.findall("author")
175-
],
176-
"year": int(year),
177-
"venue": venue_of(entry),
178-
"type": "preprint" if is_preprint else entry.tag,
179-
"url": f"https://arxiv.org/abs/{arxiv_id}" if arxiv_id else best_link(entry),
180-
"dblp": dblp_link(entry),
181-
}
182-
(preprints if is_preprint else peer_reviewed).append(publication)
183+
for row in payload["results"]["bindings"]:
184+
def value(name: str) -> str:
185+
return row.get(name, {}).get("value", "").strip()
186+
187+
title = clean_title(value("title"))
188+
key = value("pub").removeprefix(REC_PREFIX)
189+
if not title or not key:
190+
continue
191+
192+
rdf_type = value("type").rsplit("#", 1)[-1]
193+
is_thesis = value("bibtex").endswith("Phdthesis")
194+
is_preprint = rdf_type == "Informal"
195+
kind = {
196+
"Inproceedings": "inproceedings",
197+
"Article": "article",
198+
"Informal": "preprint",
199+
"Book": "phdthesis" if is_thesis else "book",
200+
}.get(rdf_type, "inproceedings")
201+
202+
# CoRR records carry the arXiv id in the journal volume ("abs/2601.14039").
203+
arxiv_id = value("volume").removeprefix("abs/") if is_preprint else ""
204+
205+
if is_preprint:
206+
venue = "arXiv"
207+
elif is_thesis:
208+
venue = value("school")
209+
elif kind == "book":
210+
# Books read better with their publisher: "Cognitive Technologies, Springer".
211+
venue = ", ".join(p for p in (value("venue"), value("publisher")) if p)
212+
else:
213+
venue = VOLUME_SUFFIX.sub("", value("venue"))
214+
215+
# DBLP's TOC page plus the record key is the human-facing record link.
216+
toc = value("tocpage")
217+
dblp_link = f"{toc}.html#{key.rsplit('/', 1)[-1]}" if toc else ""
218+
219+
if arxiv_id:
220+
url = f"https://arxiv.org/abs/{arxiv_id}"
221+
else:
222+
url = value("doi") or value("docpage") or dblp_link
223+
224+
publication = {
225+
"key": key,
226+
"title": title,
227+
"authors": authors_of(value("authors")),
228+
"year": int(value("year")),
229+
"venue": venue,
230+
"type": kind,
231+
"url": url,
232+
"dblp": dblp_link,
233+
}
234+
(preprints if is_preprint else peer_reviewed).append(publication)
183235

184236
# A preprint that later appeared at a venue would otherwise be listed twice.
185237
published_titles = {normalize_title(p["title"]) for p in peer_reviewed}
@@ -304,7 +356,7 @@ def to_yaml(publications: list[dict]) -> str:
304356
"""Emit YAML by hand so the field order and quoting stay diff-friendly."""
305357
lines = [
306358
"# Generated by scripts/fetch_publications.py from DBLP — refreshed weekly.",
307-
f"# Source: https://dblp.org/pid/{DBLP_PID}.html (papers from {START_YEAR} onwards)",
359+
f"# Source: sparql.dblp.org, author pid {DBLP_PID} (papers from {START_YEAR} onwards)",
308360
"#",
309361
"# Hand edits: set `manual: true` on an entry to freeze it (never overwritten,",
310362
"# and kept even if DBLP drops it). `oa_url:` is preserved on every refresh",
@@ -330,8 +382,8 @@ def to_yaml(publications: list[dict]) -> str:
330382

331383

332384
def main() -> int:
333-
print(f"Fetching {DBLP_URL}")
334-
publications = merge(parse(fetch(DBLP_URL)), load_existing(OUTPUT))
385+
print(f"Querying {SPARQL_URL} for pid {DBLP_PID}")
386+
publications = merge(parse(fetch()), load_existing(OUTPUT))
335387
if not publications:
336388
print("Refusing to write an empty publication list.", file=sys.stderr)
337389
return 1

0 commit comments

Comments
 (0)