Coverage for src/crawler/by_source/ams_crawler.py: 14%

143 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1import html 

2import json 

3import os 

4from urllib.parse import urljoin 

5from uuid import uuid4 

6 

7from bs4 import BeautifulSoup, Tag 

8from opentelemetry import trace 

9from ptf.cmds.xml.ckeditor.ckeditor_parser import CkeditorParser 

10from ptf.cmds.xml.ckeditor.utils import get_abstract_xml 

11from ptf.model_data import ( 

12 create_abstract, 

13 create_articledata, 

14 create_contributor, 

15 create_extid, 

16 create_subj, 

17) 

18from ptf.utils import execute_cmd 

19 

20from crawler.abstract_crawlers.matching_crawler import MatchingCrawler 

21from crawler.cmds.mixed_citation import ExtLinkXml, MixedCitation 

22from crawler.tests.data_generation.decorators import skip_generation 

23from crawler.utils import add_pdf_link_to_xarticle, cleanup_str 

24 

25 

26class AmsCrawler(MatchingCrawler): 

27 source_name = "American Mathematical Society" 

28 source_domain = "AMS" 

29 source_website = "https://www.ams.org/" 

30 tracer = trace.get_tracer(__name__) 

31 

32 @classmethod 

33 def get_view_id(cls): 

34 return "AMS" 

35 

36 @skip_generation 

37 def parse_collection_content(self, content): 

38 xissues = [] 

39 soup = BeautifulSoup(content, "html.parser") 

40 issues_data_tag = soup.select_one( 

41 ".container main[role='main'] script[type='text/javascript']:not([src])" 

42 ) 

43 data = json.loads(self.get_col_issues(issues_data_tag.text)) 

44 issues = data["issues"] 

45 self.group_by_year = data["group_by_year"] == "Y" 

46 self.ams_code = data["ams_code"].lower() 

47 for i in issues: 

48 number = i.get("IssueNumber", None) 

49 if number: 

50 number = str(number) 

51 if self.group_by_year: 

52 number = None 

53 # For AMS, xissue.url is NOT a real URL, but the AMS issue ID 

54 # Issue data is fetched from an API and thus every issue url is the same 

55 xissues.append( 

56 self.create_xissue( 

57 str(i["IssueId"]), 

58 int(i["Year"]), 

59 str(i["Volume"]), 

60 number, 

61 ) 

62 ) 

63 

64 if self.group_by_year: 

65 # We take only the first issue advertised by the website 

66 # All ignored issues will be present inside the API on the next step anyways 

67 years = {} 

68 for i in xissues: 

69 if i.fyear not in years: 

70 years[i.fyear] = [] 

71 years[i.fyear].append(i) 

72 

73 xissues = [y[0] for y in years.values()] 

74 return xissues 

75 

76 def get_col_issues(self, input: str): 

77 """ 

78 AMS Issues are listed inside an inline js script 

79 We have to spawn a nodejs subprocess to convert javascript into json""" 

80 

81 filename = "/tmp/crawler/puppeteer/" + str(uuid4()) 

82 filename_out = filename + "-out" 

83 os.makedirs(os.path.dirname(filename), exist_ok=True) 

84 with open(filename, "w") as file: 

85 file.write(input) 

86 

87 content = None 

88 attempt = 0 

89 while not content and attempt < 3: 

90 attempt += 1 

91 cmd = f"{os.path.dirname(os.path.realpath(__file__))}/ams_crawler_col.js -f {filename} -o {filename_out}" 

92 execute_cmd(cmd) 

93 

94 if os.path.isfile(filename_out): 

95 with open(filename_out) as file_: 

96 content = file_.read() 

97 

98 os.remove(filename) 

99 os.remove(filename_out) 

100 

101 if not content: 

102 raise ValueError("Couldn't parse collection content") 

103 return content 

104 

105 def download_issue_summary(self, issue_id): 

106 response = self.session.post( 

107 "https://pubs.ams.org/product/GetJournalIssueDetail", 

108 data={"productCode": self.ams_code, "issueId": issue_id}, 

109 headers={ 

110 "User-Agent": "Mozilla/5.0 (X11; Linux x86_64; rv:149.0) Gecko/20100101 Firefox/149.0" 

111 }, 

112 ) 

113 return response.text 

114 

115 def start_process_issue(self, xissue): 

116 issue_url = xissue.url 

117 if not issue_url: 

118 raise ValueError("Issue does not have an URL") 

119 content = self.download_issue_summary(issue_url) 

120 # API response is somehow a list of issues 

121 # Currently CAMS somehow puts every article inside a different issue in the list... 

122 articles = [] 

123 

124 if self.group_by_year: 

125 for issue in json.loads(content): 

126 articles.extend(issue["Articles"]) 

127 else: 

128 issue_json = next(i for i in json.loads(content) if str(i["IssueId"]) == xissue.url) 

129 articles = issue_json["Articles"] 

130 

131 with self.tracer.start_as_current_span("parse_issue_content"): 

132 self.parse_ams_issue_content(articles, xissue) 

133 

134 def parse_ams_issue_content(self, articles: list[dict], xissue): 

135 for index, article_dict in enumerate(articles): 

136 xarticle = create_articledata() 

137 xarticle.title_tex = article_dict["Title"] 

138 # ... 

139 # https://pubs.ams.org/mcom/2000-69-231/S0025-5718-00-01249-7 

140 if article_dict["DOI"] != "DOI_PREFIX_HERE_S0025-5718-00-01249-7": 

141 xarticle.doi = article_dict["DOI"] 

142 

143 xarticle.pid = f"a_{index}" 

144 xarticle.fpage = str(article_dict["StartPage"]) 

145 xarticle.lpage = str(article_dict["EndPage"]) 

146 xarticle.date_published = article_dict["PostDate"] 

147 

148 if article_dict["DocumentType"] == "BOOKREV": 

149 if xarticle.title_tex == "": 

150 book_title = article_dict["BookReviews"][0]["Title"] 

151 xarticle.title_tex = "Book review: " + book_title 

152 if article_dict["PrimaryMsc"] is not None: 

153 for msc in article_dict["PrimaryMsc"].split(", "): 

154 xarticle.kwds.append(create_subj(type="msc", value=cleanup_str(msc))) 

155 if article_dict["SecondaryMsc"] is not None: 

156 for msc in article_dict["SecondaryMsc"].split(", "): 

157 xarticle.kwds.append(create_subj(type="msc", value=cleanup_str(msc))) 

158 

159 ckeditor_data = CkeditorParser( 

160 html_value=article_dict["Abstract"], 

161 mml_formulas="", 

162 ) 

163 abstract = create_abstract( 

164 lang="en", 

165 value_xml=get_abstract_xml(ckeditor_data.value_xml, lang="en"), 

166 value_tex=ckeditor_data.value_tex, 

167 value_html=ckeditor_data.value_html, 

168 ) 

169 xarticle.abstracts.append(abstract) 

170 

171 # TODO : EnhancedReferences 

172 # TODO : UnenhancedReferences 

173 # TODO : BibliographicInfo 

174 

175 add_pdf_link_to_xarticle( 

176 xarticle, 

177 urljoin("https://www.ams.org/journals/", self.ams_code + article_dict["PdfUrl"]), 

178 ) 

179 xarticle.url = urljoin( 

180 self.collection_url, 

181 self.ams_code + "/" + article_dict["IssueDirectory"] + "/" + article_dict["PII"], 

182 ) 

183 if article_dict["MRNumber"]: 

184 extid = create_extid("mr-item-id", article_dict["MRNumber"]) 

185 xarticle.extids.append(extid) 

186 

187 for author in article_dict["Authors"]: 

188 # TODO : AMS Provides Firstname/MiddleName/LastName but we do not have Middlename fields 

189 # How should we proceed about that ? 

190 xarticle.contributors.append( 

191 create_contributor( 

192 role="author", 

193 string_name=html.unescape(author["FullName"]), 

194 email=html.unescape(author["Email"] or ""), 

195 addresses=[html.unescape(author["Affiliation"] or "")], 

196 ) 

197 ) 

198 

199 soup = BeautifulSoup(article_dict["EnhancedReferences"], "html5lib") 

200 refs = soup.select("ul > li") 

201 for ref in refs: 

202 xarticle.bibitems.append(self.parse_ref(ref)) 

203 

204 xissue.articles.append(xarticle) 

205 

206 def parse_ref(self, ref: "Tag"): 

207 citation_builder = MixedCitation() 

208 for el in ref.children: 

209 if isinstance(el, str): 

210 if el in [", DOI ", " DOI ", "DOI"]: 

211 continue 

212 citation_builder.elements.append(el) 

213 continue 

214 if isinstance(el, Tag): 

215 if el.name == "a": 

216 if el.text.startswith("10."): 

217 extlink = ExtLinkXml(urljoin("https://doi.org/", el.text)) 

218 citation_builder.elements.append(extlink) 

219 el.decompose() 

220 continue 

221 

222 href = el.get("href") 

223 if not isinstance(href, str): 

224 continue 

225 if href.startswith("https://mathscinet.ams.org/mathscinet-getitem"): 

226 extlink = ExtLinkXml(href) 

227 citation_builder.elements.append(extlink) 

228 el.decompose() 

229 continue 

230 citation_builder.elements.append(el.get_text()) 

231 return citation_builder.get_jats_ref() 

232 

233 # def parse_ref(self, ref: "Tag"): 

234 # citation_builder = MixedCitation() 

235 # # Everything behind the title should be authors 

236 # title_element = ref.select_one("em") 

237 # if title_element: 

238 # authors = list(title_element.previous_siblings) 

239 # # if len(authors) != 1: 

240 # # self.logger.error("Could not correctly parse reference. Fallback to text") 

241 # # citation_builder.elements.append(ref.get_text()) 

242 # # return citation_builder.get_jats_ref() 

243 # # Temporary fix : structured bibitems parsing is sometimes incorrect. 

244 # # Better have no data than incorrect data (?) 

245 # for el in authors: 

246 # citation_builder.elements.append(el.get_text()) 

247 # for el in authors: 

248 # el.extract() 

249 # # authors_el = GenericRefElement() 

250 # # authors_el.name = "person-group" 

251 # # citation_builder.elements.append(authors_el) 

252 # # authors_text = authors[0].text 

253 # # if authors_text.endswith(", "): 

254 # # authors_text = authors_text.removesuffix(", ") 

255 # # authors_el.elements.append(authors_text) 

256 # # citation_builder.elements.append(", ") 

257 # # else: 

258 # # authors_el.elements.append(authors_text) 

259 

260 # article_title = MixedCitation() 

261 # article_title.name = "article-title" 

262 # citation_builder.elements.append(article_title) 

263 # article_title.elements.append(title_element.text) 

264 # title_element.decompose() 

265 

266 # # everything before a tag is text 

267 # first_link = ref.select_one("a") 

268 # if first_link: 

269 # texts = list(first_link.previous_siblings) 

270 # if len(texts) == 0: 

271 # raise ValueError("first_link previous_siblings is empty") 

272 # for el in reversed(texts): 

273 # citation_builder.elements.append(el.get_text().removesuffix(", Preprint, arXiv:")) 

274 # el.extract() 

275 

276 # for link in ref.select("a"): 

277 # url = link.get("href") 

278 # if not isinstance(url, str): 

279 # raise ValueError("Citation extlink does not have a valid url") 

280 # reflink = ExtLinkXml(url) 

281 # citation_builder.elements.append(reflink) 

282 # else: 

283 # citation_builder.elements.append(ref.get_text()) 

284 

285 # return citation_builder.get_jats_ref()