Coverage for src/crawler/by_source/ams_crawler.py: 14%
143 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import html
2import json
3import os
4from urllib.parse import urljoin
5from uuid import uuid4
7from bs4 import BeautifulSoup, Tag
8from opentelemetry import trace
9from ptf.cmds.xml.ckeditor.ckeditor_parser import CkeditorParser
10from ptf.cmds.xml.ckeditor.utils import get_abstract_xml
11from ptf.model_data import (
12 create_abstract,
13 create_articledata,
14 create_contributor,
15 create_extid,
16 create_subj,
17)
18from ptf.utils import execute_cmd
20from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
21from crawler.cmds.mixed_citation import ExtLinkXml, MixedCitation
22from crawler.tests.data_generation.decorators import skip_generation
23from crawler.utils import add_pdf_link_to_xarticle, cleanup_str
26class AmsCrawler(MatchingCrawler):
27 source_name = "American Mathematical Society"
28 source_domain = "AMS"
29 source_website = "https://www.ams.org/"
30 tracer = trace.get_tracer(__name__)
32 @classmethod
33 def get_view_id(cls):
34 return "AMS"
36 @skip_generation
37 def parse_collection_content(self, content):
38 xissues = []
39 soup = BeautifulSoup(content, "html.parser")
40 issues_data_tag = soup.select_one(
41 ".container main[role='main'] script[type='text/javascript']:not([src])"
42 )
43 data = json.loads(self.get_col_issues(issues_data_tag.text))
44 issues = data["issues"]
45 self.group_by_year = data["group_by_year"] == "Y"
46 self.ams_code = data["ams_code"].lower()
47 for i in issues:
48 number = i.get("IssueNumber", None)
49 if number:
50 number = str(number)
51 if self.group_by_year:
52 number = None
53 # For AMS, xissue.url is NOT a real URL, but the AMS issue ID
54 # Issue data is fetched from an API and thus every issue url is the same
55 xissues.append(
56 self.create_xissue(
57 str(i["IssueId"]),
58 int(i["Year"]),
59 str(i["Volume"]),
60 number,
61 )
62 )
64 if self.group_by_year:
65 # We take only the first issue advertised by the website
66 # All ignored issues will be present inside the API on the next step anyways
67 years = {}
68 for i in xissues:
69 if i.fyear not in years:
70 years[i.fyear] = []
71 years[i.fyear].append(i)
73 xissues = [y[0] for y in years.values()]
74 return xissues
76 def get_col_issues(self, input: str):
77 """
78 AMS Issues are listed inside an inline js script
79 We have to spawn a nodejs subprocess to convert javascript into json"""
81 filename = "/tmp/crawler/puppeteer/" + str(uuid4())
82 filename_out = filename + "-out"
83 os.makedirs(os.path.dirname(filename), exist_ok=True)
84 with open(filename, "w") as file:
85 file.write(input)
87 content = None
88 attempt = 0
89 while not content and attempt < 3:
90 attempt += 1
91 cmd = f"{os.path.dirname(os.path.realpath(__file__))}/ams_crawler_col.js -f {filename} -o {filename_out}"
92 execute_cmd(cmd)
94 if os.path.isfile(filename_out):
95 with open(filename_out) as file_:
96 content = file_.read()
98 os.remove(filename)
99 os.remove(filename_out)
101 if not content:
102 raise ValueError("Couldn't parse collection content")
103 return content
105 def download_issue_summary(self, issue_id):
106 response = self.session.post(
107 "https://pubs.ams.org/product/GetJournalIssueDetail",
108 data={"productCode": self.ams_code, "issueId": issue_id},
109 headers={
110 "User-Agent": "Mozilla/5.0 (X11; Linux x86_64; rv:149.0) Gecko/20100101 Firefox/149.0"
111 },
112 )
113 return response.text
115 def start_process_issue(self, xissue):
116 issue_url = xissue.url
117 if not issue_url:
118 raise ValueError("Issue does not have an URL")
119 content = self.download_issue_summary(issue_url)
120 # API response is somehow a list of issues
121 # Currently CAMS somehow puts every article inside a different issue in the list...
122 articles = []
124 if self.group_by_year:
125 for issue in json.loads(content):
126 articles.extend(issue["Articles"])
127 else:
128 issue_json = next(i for i in json.loads(content) if str(i["IssueId"]) == xissue.url)
129 articles = issue_json["Articles"]
131 with self.tracer.start_as_current_span("parse_issue_content"):
132 self.parse_ams_issue_content(articles, xissue)
134 def parse_ams_issue_content(self, articles: list[dict], xissue):
135 for index, article_dict in enumerate(articles):
136 xarticle = create_articledata()
137 xarticle.title_tex = article_dict["Title"]
138 # ...
139 # https://pubs.ams.org/mcom/2000-69-231/S0025-5718-00-01249-7
140 if article_dict["DOI"] != "DOI_PREFIX_HERE_S0025-5718-00-01249-7":
141 xarticle.doi = article_dict["DOI"]
143 xarticle.pid = f"a_{index}"
144 xarticle.fpage = str(article_dict["StartPage"])
145 xarticle.lpage = str(article_dict["EndPage"])
146 xarticle.date_published = article_dict["PostDate"]
148 if article_dict["DocumentType"] == "BOOKREV":
149 if xarticle.title_tex == "":
150 book_title = article_dict["BookReviews"][0]["Title"]
151 xarticle.title_tex = "Book review: " + book_title
152 if article_dict["PrimaryMsc"] is not None:
153 for msc in article_dict["PrimaryMsc"].split(", "):
154 xarticle.kwds.append(create_subj(type="msc", value=cleanup_str(msc)))
155 if article_dict["SecondaryMsc"] is not None:
156 for msc in article_dict["SecondaryMsc"].split(", "):
157 xarticle.kwds.append(create_subj(type="msc", value=cleanup_str(msc)))
159 ckeditor_data = CkeditorParser(
160 html_value=article_dict["Abstract"],
161 mml_formulas="",
162 )
163 abstract = create_abstract(
164 lang="en",
165 value_xml=get_abstract_xml(ckeditor_data.value_xml, lang="en"),
166 value_tex=ckeditor_data.value_tex,
167 value_html=ckeditor_data.value_html,
168 )
169 xarticle.abstracts.append(abstract)
171 # TODO : EnhancedReferences
172 # TODO : UnenhancedReferences
173 # TODO : BibliographicInfo
175 add_pdf_link_to_xarticle(
176 xarticle,
177 urljoin("https://www.ams.org/journals/", self.ams_code + article_dict["PdfUrl"]),
178 )
179 xarticle.url = urljoin(
180 self.collection_url,
181 self.ams_code + "/" + article_dict["IssueDirectory"] + "/" + article_dict["PII"],
182 )
183 if article_dict["MRNumber"]:
184 extid = create_extid("mr-item-id", article_dict["MRNumber"])
185 xarticle.extids.append(extid)
187 for author in article_dict["Authors"]:
188 # TODO : AMS Provides Firstname/MiddleName/LastName but we do not have Middlename fields
189 # How should we proceed about that ?
190 xarticle.contributors.append(
191 create_contributor(
192 role="author",
193 string_name=html.unescape(author["FullName"]),
194 email=html.unescape(author["Email"] or ""),
195 addresses=[html.unescape(author["Affiliation"] or "")],
196 )
197 )
199 soup = BeautifulSoup(article_dict["EnhancedReferences"], "html5lib")
200 refs = soup.select("ul > li")
201 for ref in refs:
202 xarticle.bibitems.append(self.parse_ref(ref))
204 xissue.articles.append(xarticle)
206 def parse_ref(self, ref: "Tag"):
207 citation_builder = MixedCitation()
208 for el in ref.children:
209 if isinstance(el, str):
210 if el in [", DOI ", " DOI ", "DOI"]:
211 continue
212 citation_builder.elements.append(el)
213 continue
214 if isinstance(el, Tag):
215 if el.name == "a":
216 if el.text.startswith("10."):
217 extlink = ExtLinkXml(urljoin("https://doi.org/", el.text))
218 citation_builder.elements.append(extlink)
219 el.decompose()
220 continue
222 href = el.get("href")
223 if not isinstance(href, str):
224 continue
225 if href.startswith("https://mathscinet.ams.org/mathscinet-getitem"):
226 extlink = ExtLinkXml(href)
227 citation_builder.elements.append(extlink)
228 el.decompose()
229 continue
230 citation_builder.elements.append(el.get_text())
231 return citation_builder.get_jats_ref()
233 # def parse_ref(self, ref: "Tag"):
234 # citation_builder = MixedCitation()
235 # # Everything behind the title should be authors
236 # title_element = ref.select_one("em")
237 # if title_element:
238 # authors = list(title_element.previous_siblings)
239 # # if len(authors) != 1:
240 # # self.logger.error("Could not correctly parse reference. Fallback to text")
241 # # citation_builder.elements.append(ref.get_text())
242 # # return citation_builder.get_jats_ref()
243 # # Temporary fix : structured bibitems parsing is sometimes incorrect.
244 # # Better have no data than incorrect data (?)
245 # for el in authors:
246 # citation_builder.elements.append(el.get_text())
247 # for el in authors:
248 # el.extract()
249 # # authors_el = GenericRefElement()
250 # # authors_el.name = "person-group"
251 # # citation_builder.elements.append(authors_el)
252 # # authors_text = authors[0].text
253 # # if authors_text.endswith(", "):
254 # # authors_text = authors_text.removesuffix(", ")
255 # # authors_el.elements.append(authors_text)
256 # # citation_builder.elements.append(", ")
257 # # else:
258 # # authors_el.elements.append(authors_text)
260 # article_title = MixedCitation()
261 # article_title.name = "article-title"
262 # citation_builder.elements.append(article_title)
263 # article_title.elements.append(title_element.text)
264 # title_element.decompose()
266 # # everything before a tag is text
267 # first_link = ref.select_one("a")
268 # if first_link:
269 # texts = list(first_link.previous_siblings)
270 # if len(texts) == 0:
271 # raise ValueError("first_link previous_siblings is empty")
272 # for el in reversed(texts):
273 # citation_builder.elements.append(el.get_text().removesuffix(", Preprint, arXiv:"))
274 # el.extract()
276 # for link in ref.select("a"):
277 # url = link.get("href")
278 # if not isinstance(url, str):
279 # raise ValueError("Citation extlink does not have a valid url")
280 # reflink = ExtLinkXml(url)
281 # citation_builder.elements.append(reflink)
282 # else:
283 # citation_builder.elements.append(ref.get_text())
285 # return citation_builder.get_jats_ref()