Coverage for src/crawler/by_source/dmlpl_crawler.py: 9%

158 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1import json 

2from urllib import parse 

3 

4from bs4 import BeautifulSoup, Tag 

5from ptf.external.session import get_session 

6from ptf.model_data import create_abstract, create_articledata, create_contributor, create_subj 

7 

8from crawler.abstract_crawlers.matching_crawler import MatchingCrawler 

9from crawler.by_source.lofpl_crawler import LofplCrawler 

10from crawler.crawler_utils import set_pages 

11from crawler.utils import add_pdf_link_to_xarticle, cleanup_str 

12 

13 

14class DmlplCrawler(MatchingCrawler): 

15 source_name = "The Polish Digital Mathematics Library" 

16 source_domain = "DMLPL" 

17 source_website = "http://pldml.icm.edu.pl/pldml" 

18 

19 # HACK : Workaround for tests (monkeypatching) 

20 # We store the class here, so we can monkeypatch it when running tests 

21 subCrawlers = {LofplCrawler: None} 

22 

23 def parse_collection_content(self, content): 

24 """ 

25 Parse the HTML page of Annals of Math and returns a list of xissue. 

26 Each xissue has its pid/volume/number/year metadata + its url 

27 """ 

28 issues = [] 

29 data = json.loads(content) 

30 for entry in data: 

31 link = self.source_website + "/tree/hierarchy.action" 

32 params = {"root": entry["id"]} 

33 link += "?" + parse.urlencode(params) 

34 

35 text: str = entry["text"] 

36 if not text.startswith("tom/rocznik"): 

37 raise ValueError( 

38 'Cannot parse Collection : couldn\'t find "tom/rocznik" at the start of the string' 

39 ) 

40 soup = BeautifulSoup(text, "html.parser") 

41 a_tags = soup.select("a") 

42 if len(a_tags) < 2: 

43 raise ValueError("Cannot parse Collection : couldn't find volume information") 

44 volume = a_tags[0].text 

45 year = int(a_tags[1].text) 

46 

47 issues.extend(self.parse_dmlpl_volume_content(link, year, volume)) 

48 return issues 

49 

50 def parse_dmlpl_volume_content(self, link: str, year: int, volume: str): 

51 content = self.download_file(link) 

52 has_articles = False 

53 issues = [] 

54 data = json.loads(content) 

55 for entry in data: 

56 entry_link = self.source_website + "/tree/hierarchy.action" 

57 params = {"root": entry["id"]} 

58 entry_link += "?" + parse.urlencode(params) 

59 

60 number = None 

61 text: str = entry["text"] 

62 if text.startswith("numer"): 

63 soup = BeautifulSoup(text, "html.parser") 

64 a_tag = soup.select_one("a") 

65 if not a_tag: 

66 raise ValueError("Cannot parse Collection : couldn't find issue information") 

67 number = a_tag.text.replace(" ", "_") 

68 issues.append(self.create_xissue(entry_link, year, volume, number)) 

69 elif text.startswith("artykuł"): 

70 has_articles = True 

71 

72 if has_articles: 

73 issues.append(self.create_xissue(link, year, volume)) 

74 

75 return issues 

76 

77 def parse_issue_content(self, content, xissue): 

78 data = json.loads(content) 

79 for index, entry in enumerate(data): 

80 xarticle = create_articledata() 

81 xarticle.pid = "a" + str(index) 

82 xarticle.url = self.source_website + "/element/" + entry["id"] 

83 xissue.articles.append(xarticle) 

84 

85 # IDEA : manually following redirections would allow us to get the redirection URL without the body (for bibliotekanauki) 

86 def crawl_article(self, xarticle, xissue): 

87 parsed_xarticle = xarticle 

88 if hasattr(xarticle, "url") and xarticle.url: 

89 response = get_session().head( 

90 xarticle.url, 

91 ) 

92 # Crawl using LOFPL if detected 

93 if response.url.startswith("https://bibliotekanauki.pl"): 

94 xarticle.url = response.url.replace( 

95 "https://bibliotekanauki.pl", "https://bibliotekanauki.pl/api" 

96 ) 

97 targetCrawler = self.subCrawlers[LofplCrawler] 

98 if targetCrawler is None: 

99 raise ValueError("Crawler incorrectly initialized") 

100 parsed_xarticle = targetCrawler.crawl_article(xarticle, xissue) 

101 elif response.url.startswith("http://pldml.icm.edu.pl"): 

102 parsed_xarticle = super().crawl_article(xarticle, xissue) 

103 else: 

104 raise NotImplementedError 

105 

106 if not parsed_xarticle: 

107 raise ValueError("Couldn't crawl article") 

108 # The article title may have formulas surrounded with '$' 

109 return self.process_article_metadata(parsed_xarticle) 

110 

111 def parse_dmlpl_generic_page(self, content: str): 

112 soup = BeautifulSoup(content, "html.parser") 

113 main = soup.select_one("div.details-content") 

114 if not main: 

115 raise ValueError("Cannot parse article : main div not found") 

116 

117 sections = main.select("div.row") 

118 sections_dict: dict[str, Tag] = {} 

119 for s in sections: 

120 row_label = s.select_one("div.row-label") 

121 if not row_label: 

122 raise ValueError("Cannot parse article : row label not found") 

123 tag = s.select_one("div.row-desc") 

124 if tag: 

125 sections_dict[row_label.text] = tag 

126 

127 return sections_dict 

128 

129 def parse_article_content(self, content, xissue, xarticle, url): 

130 sections_dict = self.parse_dmlpl_generic_page(content) 

131 

132 xarticle.title_tex = cleanup_str(sections_dict["Tytuł artykułu"].text) 

133 # DOI 

134 if "Identyfikatory" in sections_dict: 

135 doi_tag = sections_dict["Identyfikatory"].select_one("a[href*='doi.org']") 

136 xarticle.doi = cleanup_str(doi_tag.text) 

137 

138 # Author 

139 for a_tag in sections_dict["Autorzy"].select("a"): 

140 href = a_tag.get("href") 

141 if not isinstance(href, str): 

142 raise ValueError("author href is not a string") 

143 author = self.parse_author(self.download_file(self.source_website + "/" + href)) 

144 author["role"] = "author" 

145 xarticle.contributors.append(author) 

146 

147 # TODO : Contributor ? (Twórcy) 

148 

149 # PDF 

150 if "Treść / Zawartość" in sections_dict: 

151 pdf_a_tag = sections_dict["Treść / Zawartość"].select_one("a") 

152 if not pdf_a_tag: 

153 raise ValueError("Cannot find pdf for article") 

154 pdf_url = pdf_a_tag.get("href") 

155 if not isinstance(pdf_url, str): 

156 raise ValueError("Cannot parse pdf url for article") 

157 if not pdf_url.startswith("http"): 

158 pdf_url = self.source_website + "/" + pdf_url 

159 add_pdf_link_to_xarticle(xarticle, pdf_url) 

160 else: 

161 self.logger.info("PDF not found", extra={"pid": xarticle.pid}) 

162 

163 # Lang 

164 xarticle.lang = cleanup_str(sections_dict["Języki publikacji"].text.lower()) 

165 if len(xarticle.lang) > 3: 

166 if xarticle.lang == "pl fr": 

167 xarticle.lang = "pl" 

168 self.logger.info( 

169 f"[{xarticle.pid}] Patch : set article lang to 'pl' (was 'pl fr' before)", 

170 extra={"pid": xarticle.pid}, 

171 ) 

172 else: 

173 raise ValueError("Cannot parse article lang") 

174 

175 # Abstract 

176 if "Abstrakty" in sections_dict: 

177 abstract_divs = sections_dict["Abstrakty"].select("div.listing-row") 

178 for div in abstract_divs: 

179 lang = "und" 

180 lang_div = div.select_one("div.articleDetails-langCell") 

181 if lang_div: 

182 lang = cleanup_str(lang_div.text).lower() 

183 text_div = div.select_one("div.articleDetails-abstract") 

184 if not text_div: 

185 raise ValueError( 

186 "Error while parsing abstract : abstract presence detected, but abstract cannot be parsed" 

187 ) 

188 abstract_text = cleanup_str(text_div.text) 

189 if abstract_text != "-": 

190 xarticle.abstracts.append(create_abstract(value_tex=abstract_text, lang=lang)) 

191 

192 # Keywords 

193 if "Słowa kluczowe" in sections_dict: 

194 keywords_lists = sections_dict["Słowa kluczowe"].select("div.listing-row") 

195 for list in keywords_lists: 

196 lang = "und" 

197 lang_div = list.select_one("div.articleDetails-langCell") 

198 keywords_a_tags = list.select("a") 

199 for a_tag in keywords_a_tags: 

200 subject = create_subj() 

201 subject["value"] = a_tag.text 

202 subject["lang"] = lang 

203 xarticle.kwds.append(subject) 

204 # Page 

205 if "Strony" in sections_dict: 

206 set_pages(xarticle, cleanup_str(sections_dict["Strony"].text)) 

207 

208 return xarticle 

209 

210 def parse_author(self, content: str): 

211 author = create_contributor() 

212 sections_dict = self.parse_dmlpl_generic_page(content) 

213 author["last_name"] = cleanup_str(sections_dict["Nazwisko"].text) 

214 author["first_name"] = cleanup_str(sections_dict["Imię"].text) 

215 if len(author["last_name"]) == 0 or len(author["first_name"]) == 0: 

216 author["string_name"] = cleanup_str(sections_dict["Twórca"].text) 

217 return author