Coverage for src/crawler/crawler_utils.py: 17%
210 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1# This file contains utils functions related to ArticleData or IssueData parsing and population
2# Some of the functions present here were initially present in base_crawler but then moved here.
5import logging
6from collections.abc import Callable
7from email.policy import EmailPolicy
9import regex
10from bs4 import BeautifulSoup
11from langcodes import standardize_tag
12from ptf.cmds.xml.jats.builder.references import (
13 get_article_title_xml,
14 get_author_xml,
15 get_fpage_xml,
16 get_lpage_xml,
17 get_source_xml,
18 get_year_xml,
19)
20from ptf.cmds.xml.jats.jats_parser import parse_mixed_citation_into_ref
21from ptf.model_data import (
22 ArticleData,
23 ContributorDict,
24 IssueData,
25 create_abstract,
26 create_contributor,
27 create_extid,
28 create_issuedata,
29 create_publisherdata,
30)
32from crawler.types import CitationLiteral
33from crawler.utils import add_pdf_link_to_xarticle, cleanup_str
35references_mapping = {
36 "citation_title": get_article_title_xml,
37 "citation_journal_title": get_source_xml,
38 "citation_publication_date": get_year_xml,
39 "citation_firstpage": get_fpage_xml,
40 "citation_lastpage": get_lpage_xml,
41}
43logger = logging.getLogger(__name__)
46def parse_content_type_charset(content_type: str):
47 header = EmailPolicy.header_factory("content-type", content_type)
48 if "charset" in header.params:
49 return header.params.get("charset")
52def parse_meta_citation_reference(content: str, label=None):
53 categories = content.split(";")
55 if len(categories) == 1:
56 return parse_mixed_citation_into_ref(content, label=label)
58 citation_data = [c.split("=") for c in categories if "=" in c]
59 del categories
61 xml_string = ""
62 authors_parsed = False
63 authors_strings = []
64 for data in citation_data:
65 key = data[0].strip()
66 citation_content = data[1]
67 if key == "citation_author":
68 authors_strings.append(get_author_xml(template_str=citation_content))
69 continue
70 elif not authors_parsed:
71 xml_string += ", ".join(authors_strings)
72 authors_parsed = True
74 if key in references_mapping:
75 xml_string += " " + references_mapping[key](citation_content)
77 return parse_mixed_citation_into_ref(xml_string, label=label)
80def set_pages(article: ArticleData, pages: str, separator: str = "-"):
81 pages_split = pages.split(separator)
82 if len(pages_split) == 0: 82 ↛ 83line 82 didn't jump to line 83 because the condition on line 82 was never true
83 article.page_range = pages
84 if len(pages_split) > 0: 84 ↛ exitline 84 didn't return from function 'set_pages' because the condition on line 84 was always true
85 if pages[0].isnumeric(): 85 ↛ exitline 85 didn't return from function 'set_pages' because the condition on line 85 was always true
86 article.fpage = pages_split[0]
87 if (
88 len(pages_split) > 1
89 and pages_split[0] != pages_split[1]
90 and pages_split[1].isnumeric()
91 ):
92 article.lpage = pages_split[1]
95def get_issue_pid(
96 collection_id: str,
97 year: int,
98 volume_number: str | None = None,
99 issue_number: str | None = None,
100 series: str | None = None,
101):
102 # Replace any non-word character with an underscore
103 pid = f"{collection_id}_{year}"
104 if series is not None: 104 ↛ 105line 104 didn't jump to line 105 because the condition on line 104 was never true
105 pid += f"_{series}"
106 if volume_number is not None: 106 ↛ 108line 106 didn't jump to line 108 because the condition on line 106 was always true
107 pid += f"_{volume_number}"
108 if issue_number is not None: 108 ↛ 110line 108 didn't jump to line 110 because the condition on line 108 was always true
109 pid += f"_{issue_number}"
110 pid = regex.sub(r"[^a-zA-Z0-9-]+", "_", cleanup_str(pid))
111 return pid
114def create_xissue(
115 collection_id: str,
116 url: str | None,
117 year: int,
118 volume_number: str | None,
119 issue_number: str | None = "1",
120 vseries: str | None = None,
121):
122 if url is not None and url.endswith("/"): 122 ↛ 123line 122 didn't jump to line 123 because the condition on line 122 was never true
123 url = url[:-1]
124 xissue = create_issuedata()
125 xissue.url = url
127 xissue.pid = get_issue_pid(collection_id, year, volume_number, issue_number, vseries)
129 xissue.fyear = year
131 if volume_number is not None: 131 ↛ 134line 131 didn't jump to line 134 because the condition on line 131 was always true
132 xissue.volume = regex.sub(r"[^a-zA-Z0-9-]+", "_", volume_number)
134 if issue_number is not None: 134 ↛ 137line 134 didn't jump to line 137 because the condition on line 134 was always true
135 xissue.number = issue_number.replace(",", "-")
137 if vseries is not None: 137 ↛ 138line 137 didn't jump to line 138 because the condition on line 137 was never true
138 xissue.vseries = vseries
139 return xissue
142def get_metadata_using_citation_meta(
143 xarticle: ArticleData,
144 xissue: IssueData,
145 soup: BeautifulSoup,
146 what: list[CitationLiteral] = [],
147 detect_language_fct: Callable[[str, ArticleData], str] | None = None,
148):
149 """
150 :param xarticle: the xarticle that will collect the metadata
151 :param xissue: the xissue that will collect the publisher
152 :param soup: the BeautifulSoup object of tha article page
153 :param what: list of citation_ items to collect.
154 :return: None. The given article is modified
155 """
157 if "title" in what:
158 # TITLE
159 citation_title_node = soup.select_one("meta[name='citation_title']")
160 if citation_title_node:
161 title = citation_title_node.get("content")
162 if isinstance(title, str):
163 xarticle.title_tex = title
165 if "author" in what:
166 # AUTHORS
167 citation_author_nodes = soup.select("meta[name^='citation_author']")
168 current_author: ContributorDict | None = None
169 for citation_author_node in citation_author_nodes:
170 if citation_author_node.get("name") == "citation_author":
171 text_author = citation_author_node.get("content")
172 if not isinstance(text_author, str):
173 raise ValueError("Cannot parse author")
174 if text_author == "":
175 current_author = None
176 continue
177 current_author = create_contributor(role="author", string_name=text_author)
178 xarticle.contributors.append(current_author)
179 continue
180 if current_author is None:
181 logger.warning("Couldn't parse citation author")
182 continue
183 if citation_author_node.get("name") == "citation_author_institution":
184 text_institution = citation_author_node.get("content")
185 if not isinstance(text_institution, str):
186 continue
187 current_author["addresses"].append(text_institution)
188 if citation_author_node.get("name") == "citation_author_ocrid":
189 text_orcid = citation_author_node.get("content")
190 if not isinstance(text_orcid, str):
191 continue
192 current_author["orcid"] = text_orcid
194 if "pdf" in what:
195 # PDF
196 citation_pdf_node = soup.select_one('meta[name="citation_pdf_url"]')
197 if citation_pdf_node:
198 pdf_url = citation_pdf_node.get("content")
199 if isinstance(pdf_url, str):
200 add_pdf_link_to_xarticle(xarticle, pdf_url)
202 if "lang" in what:
203 # LANG
204 citation_lang_node = soup.select_one("meta[name='citation_language']")
205 if citation_lang_node:
206 # TODO: check other language code
207 content_text = citation_lang_node.get("content")
208 if isinstance(content_text, str):
209 xarticle.lang = standardize_tag(content_text)
211 if "abstract" in what:
212 # ABSTRACT
213 abstract_node = soup.select_one("meta[name='citation_abstract']")
214 if abstract_node is not None:
215 abstract = abstract_node.get("content")
216 if not isinstance(abstract, str):
217 raise ValueError("Couldn't parse abstract from meta")
218 abstract = BeautifulSoup(abstract, "html.parser").text
219 lang = abstract_node.get("lang")
220 if not isinstance(lang, str):
221 if not detect_language_fct:
222 return
223 lang = detect_language_fct(abstract, xarticle)
224 xarticle.abstracts.append(create_abstract(lang=lang, value_tex=abstract))
226 if "page" in what:
227 # PAGES
228 citation_fpage_node = soup.select_one("meta[name='citation_firstpage']")
229 if citation_fpage_node:
230 page = citation_fpage_node.get("content")
231 if isinstance(page, str):
232 page = page.split("(")[0]
233 if len(page) < 32:
234 xarticle.fpage = page
236 citation_lpage_node = soup.select_one("meta[name='citation_lastpage']")
237 if citation_lpage_node:
238 page = citation_lpage_node.get("content")
239 if isinstance(page, str):
240 page = page.split("(")[0]
241 if len(page) < 32:
242 xarticle.lpage = page
244 if "doi" in what:
245 # DOI
246 citation_doi_node = soup.select_one("meta[name='citation_doi']")
247 if citation_doi_node:
248 doi = citation_doi_node.get("content")
249 if isinstance(doi, str):
250 doi = doi.strip()
251 pos = doi.find("10.")
252 if pos > 0:
253 doi = doi[pos:]
254 xarticle.doi = doi
256 if "mr" in what:
257 # MR
258 citation_mr_node = soup.select_one("meta[name='citation_mr']")
259 if citation_mr_node:
260 mr = citation_mr_node.get("content")
261 if isinstance(mr, str):
262 mr = mr.strip()
263 if mr.find("MR") == 0:
264 mr = mr[2:]
265 extid = create_extid("mr-item-id", mr)
266 xarticle.extids.append(extid)
268 if "zbl" in what:
269 # ZBL
270 citation_zbl_node = soup.select_one("meta[name='citation_zbl']")
271 if citation_zbl_node:
272 zbl = citation_zbl_node.get("content")
273 if isinstance(zbl, str):
274 zbl = zbl.strip()
275 if zbl.find("Zbl") == 0:
276 zbl = zbl[3:].strip()
277 extid = create_extid("zbl-item-id", zbl)
278 xarticle.extids.append(extid)
280 if "publisher" in what:
281 # PUBLISHER
282 citation_publisher_node = soup.select_one("meta[name='citation_publisher']")
283 if citation_publisher_node:
284 pub = citation_publisher_node.get("content")
285 if isinstance(pub, str):
286 pub = pub.strip()
287 if pub != "":
288 xpub = create_publisherdata()
289 xpub.name = pub
290 xissue.publisher = xpub
292 if "keywords" in what:
293 # KEYWORDS
294 citation_kwd_nodes = soup.select("meta[name='citation_keywords']")
295 for kwd_node in citation_kwd_nodes:
296 kwds = kwd_node.get("content")
297 if isinstance(kwds, str):
298 kwds = kwds.split(",")
299 for kwd in kwds:
300 if kwd == "":
301 continue
302 kwd = kwd.strip()
303 xarticle.kwds.append({"type": "", "lang": xarticle.lang, "value": kwd})
305 if "references" in what:
306 citation_references = soup.select("meta[name='citation_reference']")
307 for index, tag in enumerate(citation_references):
308 content = tag.get("content")
309 if not isinstance(content, str):
310 raise ValueError("Cannot parse citation_reference meta")
311 label = str(index + 1)
312 if regex.match(r"^\[\d+\].*", content):
313 label = None
314 xarticle.bibitems.append(parse_meta_citation_reference(content, label))
317def article_has_pdf(art: ArticleData | IssueData):
318 return next((link for link in art.ext_links if link["rel"] == "article-pdf"), None) is not None
321def article_has_source(art: ArticleData | IssueData):
322 return (
323 next(
324 (e_link for e_link in art.ext_links if e_link["rel"] == "source"),
325 None,
326 )
327 is not None
328 )