diff --git a/bin/issue_to_bibtex.py b/bin/issue_to_bibtex.py index 48e03f8a..f25f69ca 100644 --- a/bin/issue_to_bibtex.py +++ b/bin/issue_to_bibtex.py @@ -11,7 +11,7 @@ from bibtexparser.bibdatabase import BibDatabase from bibtexparser.bwriter import BibTexWriter -from utils import fixBadBibFormat +from utils import fixBadBibFormat, getArxivInfoFromPage try: @@ -38,17 +38,28 @@ arxiv_id = url.split('/')[-1] found, items = get_arxiv_info(arxiv_id, field='id') + if not found and re.search(r'v\d+$', arxiv_id) is not None: + found, items = get_arxiv_info(re.sub(r'v\d+$', '', arxiv_id), field='id') + if not found: + try: + item = getArxivInfoFromPage(url, arxiv_id) + except requests.RequestException: + item = None + if item is not None: + found = True + items = [item] if not found: print(f"I could not find arXiv infos for {url}, maybe try without the version suffix 'v[.]'") continue if len(items) > 1: print(f'I got more than one item back from arXiv for {url}. See what I got:\n' + - '\n'.join([item.title for item in items]) + '\n' + + '\n'.join([item['title'] for item in items]) + '\n' + f'I am taking the first one only, just FYI. I hope this is the correct one..\n\n') - url = items[0].link - title = items[0].title.replace('\n', '').replace(' ', ' ') - year = items[0]['published'].split('-')[0] - authors = items[0].authors + item = items[0] + url = item['link'] + title = item['title'].replace('\n', '').replace(' ', ' ') + year = item['published'].split('-')[0] + authors = item['authors'] if len(authors) > 1: first_author = authors[0]["name"].split(" ") authors = " and ".join([author["name"] for author in authors]) @@ -78,8 +89,9 @@ id = id_orig + letters[i] i += 1 - howpublished = "arXiv:" + items[0]["id"].split('/')[-1] + \ - " [" + items[0]["arxiv_primary_category"]["term"] + "]" + howpublished = "arXiv:" + item["id"].split('/')[-1] + if item.get("arxiv_primary_category", {}).get("term"): + howpublished += " [" + item["arxiv_primary_category"]["term"] + "]" if not duplicate: bib_db = BibDatabase() @@ -92,7 +104,7 @@ "howpublished": howpublished, "url": url, "year": year, - "abstract": items[0]['summary'].replace('\n', ' ').replace(' ', ' '), + "abstract": item['summary'].replace('\n', ' ').replace(' ', ' '), } ] else: diff --git a/bin/test_utils.py b/bin/test_utils.py new file mode 100644 index 00000000..9304cd26 --- /dev/null +++ b/bin/test_utils.py @@ -0,0 +1,37 @@ +import unittest + +from utils import parseArxivPage + + +class ParseArxivPageTests(unittest.TestCase): + def test_parse_arxiv_page_extracts_metadata(self): + html = """ + + + + + + + + + + + + """ + + entry = parseArxivPage(html, "https://arxiv.org/abs/2609.24434", "2609.24434") + + self.assertEqual(entry["title"], "A parallel-in-time paper") + self.assertEqual(entry["published"], "2026-09-30") + self.assertEqual(entry["summary"], "An abstract with extra spacing.") + self.assertEqual(entry["link"], "https://arxiv.org/abs/2609.24434") + self.assertEqual(entry["id"], "https://arxiv.org/abs/2609.24434") + self.assertEqual(entry["arxiv_primary_category"]["term"], "math.NA") + self.assertEqual( + entry["authors"], + [{"name": "Jane Doe"}, {"name": "John Smith"}], + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/bin/utils.py b/bin/utils.py index 52deb1aa..2b68bc02 100644 --- a/bin/utils.py +++ b/bin/utils.py @@ -1,3 +1,26 @@ +import html +import re +from html.parser import HTMLParser + +import requests + + +class _MetaTagParser(HTMLParser): + def __init__(self): + super().__init__() + self.meta = {} + + def handle_starttag(self, tag, attrs): + if tag.lower() != "meta": + return + + attrs = dict(attrs) + name = attrs.get("name") + content = attrs.get("content") + if name is not None and content is not None: + self.meta.setdefault(name, []).append(html.unescape(content).strip()) + + def fixBadBibFormat(bibString:str)->str: """ Fixes the formatting of a BibTeX entry string by ensuring all field values are enclosed in braces. @@ -27,4 +50,49 @@ def fixBadBibFormat(bibString:str)->str: if len(item) == 2 and "{" not in item[1]: item[1] = "{"+item[1]+"}" fields[i+1] = "=".join(item) - return "@"+",".join(fields)+"}" \ No newline at end of file + return "@"+",".join(fields)+"}" + + +def parseArxivPage(html_content:str, url:str, arxiv_id:str): + parser = _MetaTagParser() + parser.feed(html_content) + meta = parser.meta + + def first(*names): + for name in names: + values = meta.get(name, []) + if len(values) > 0: + return values[0] + return "" + + title = re.sub(r'\s+', ' ', first("citation_title", "og:title")).strip() + authors = [re.sub(r'\s+', ' ', author).strip() for author in meta.get("citation_author", []) if author.strip()] + published = first("citation_date", "citation_publication_date", "citation_online_date").replace("/", "-") + summary = re.sub(r'\s+', ' ', first("citation_abstract", "description")).strip() + link = first("citation_abstract_html_url", "citation_public_url") or url + + category = "" + keywords = first("citation_keywords") + if keywords: + category_match = re.search(r'\(([^()]+)\)', keywords.split(";")[0]) + if category_match is not None: + category = category_match.group(1) + + if not title or len(authors) == 0 or not published or not summary: + return None + + return { + "title": title, + "authors": [{"name": author} for author in authors], + "published": published, + "summary": summary, + "link": link, + "id": f"https://arxiv.org/abs/{arxiv_id}", + "arxiv_primary_category": {"term": category}, + } + + +def getArxivInfoFromPage(url:str, arxiv_id:str): + req = requests.get(url, timeout=30) + req.raise_for_status() + return parseArxivPage(req.text, url, arxiv_id) \ No newline at end of file