Coverage for src/crawler/by_source/dmlpl_crawler.py: 9%
158 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-09-23 14:47 +0000
1import json
2from urllib import parse
4from bs4 import BeautifulSoup, Tag
5from ptf.external.session import get_session
6from ptf.model_data import create_abstract, create_articledata, create_contributor, create_subj
8from crawler.abstract_crawlers.matching_crawler import MatchingCrawler
9from crawler.by_source.lofpl_crawler import LofplCrawler
10from crawler.crawler_utils import set_pages
11from crawler.utils import add_pdf_link_to_xarticle, cleanup_str
14class DmlplCrawler(MatchingCrawler):
15 source_name = "The Polish Digital Mathematics Library"
16 source_domain = "DMLPL"
17 source_website = "http://pldml.icm.edu.pl/pldml"
19 # HACK : Workaround for tests (monkeypatching)
20 # We store the class here, so we can monkeypatch it when running tests
21 subCrawlers = {LofplCrawler: None}
23 def parse_collection_content(self, content):
24 """
25 Parse the HTML page of Annals of Math and returns a list of xissue.
26 Each xissue has its pid/volume/number/year metadata + its url
27 """
28 issues = []
29 data = json.loads(content)
30 for entry in data:
31 link = self.source_website + "/tree/hierarchy.action"
32 params = {"root": entry["id"]}
33 link += "?" + parse.urlencode(params)
35 text: str = entry["text"]
36 if not text.startswith("tom/rocznik"):
37 raise ValueError(
38 'Cannot parse Collection : couldn\'t find "tom/rocznik" at the start of the string'
39 )
40 soup = BeautifulSoup(text, "html.parser")
41 a_tags = soup.select("a")
42 if len(a_tags) < 2:
43 raise ValueError("Cannot parse Collection : couldn't find volume information")
44 volume = a_tags[0].text
45 year = int(a_tags[1].text)
47 issues.extend(self.parse_dmlpl_volume_content(link, year, volume))
48 return issues
50 def parse_dmlpl_volume_content(self, link: str, year: int, volume: str):
51 content = self.download_file(link)
52 has_articles = False
53 issues = []
54 data = json.loads(content)
55 for entry in data:
56 entry_link = self.source_website + "/tree/hierarchy.action"
57 params = {"root": entry["id"]}
58 entry_link += "?" + parse.urlencode(params)
60 number = None
61 text: str = entry["text"]
62 if text.startswith("numer"):
63 soup = BeautifulSoup(text, "html.parser")
64 a_tag = soup.select_one("a")
65 if not a_tag:
66 raise ValueError("Cannot parse Collection : couldn't find issue information")
67 number = a_tag.text.replace(" ", "_")
68 issues.append(self.create_xissue(entry_link, year, volume, number))
69 elif text.startswith("artykuł"):
70 has_articles = True
72 if has_articles:
73 issues.append(self.create_xissue(link, year, volume))
75 return issues
77 def parse_issue_content(self, content, xissue):
78 data = json.loads(content)
79 for index, entry in enumerate(data):
80 xarticle = create_articledata()
81 xarticle.pid = "a" + str(index)
82 xarticle.url = self.source_website + "/element/" + entry["id"]
83 xissue.articles.append(xarticle)
85 # IDEA : manually following redirections would allow us to get the redirection URL without the body (for bibliotekanauki)
86 def crawl_article(self, xarticle, xissue):
87 parsed_xarticle = xarticle
88 if hasattr(xarticle, "url") and xarticle.url:
89 response = get_session().head(
90 xarticle.url,
91 )
92 # Crawl using LOFPL if detected
93 if response.url.startswith("https://bibliotekanauki.pl"):
94 xarticle.url = response.url.replace(
95 "https://bibliotekanauki.pl", "https://bibliotekanauki.pl/api"
96 )
97 targetCrawler = self.subCrawlers[LofplCrawler]
98 if targetCrawler is None:
99 raise ValueError("Crawler incorrectly initialized")
100 parsed_xarticle = targetCrawler.crawl_article(xarticle, xissue)
101 elif response.url.startswith("http://pldml.icm.edu.pl"):
102 parsed_xarticle = super().crawl_article(xarticle, xissue)
103 else:
104 raise NotImplementedError
106 if not parsed_xarticle:
107 raise ValueError("Couldn't crawl article")
108 # The article title may have formulas surrounded with '$'
109 return self.process_article_metadata(parsed_xarticle)
111 def parse_dmlpl_generic_page(self, content: str):
112 soup = BeautifulSoup(content, "html.parser")
113 main = soup.select_one("div.details-content")
114 if not main:
115 raise ValueError("Cannot parse article : main div not found")
117 sections = main.select("div.row")
118 sections_dict: dict[str, Tag] = {}
119 for s in sections:
120 row_label = s.select_one("div.row-label")
121 if not row_label:
122 raise ValueError("Cannot parse article : row label not found")
123 tag = s.select_one("div.row-desc")
124 if tag:
125 sections_dict[row_label.text] = tag
127 return sections_dict
129 def parse_article_content(self, content, xissue, xarticle, url):
130 sections_dict = self.parse_dmlpl_generic_page(content)
132 xarticle.title_tex = cleanup_str(sections_dict["Tytuł artykułu"].text)
133 # DOI
134 if "Identyfikatory" in sections_dict:
135 doi_tag = sections_dict["Identyfikatory"].select_one("a[href*='doi.org']")
136 xarticle.doi = cleanup_str(doi_tag.text)
138 # Author
139 for a_tag in sections_dict["Autorzy"].select("a"):
140 href = a_tag.get("href")
141 if not isinstance(href, str):
142 raise ValueError("author href is not a string")
143 author = self.parse_author(self.download_file(self.source_website + "/" + href))
144 author["role"] = "author"
145 xarticle.contributors.append(author)
147 # TODO : Contributor ? (Twórcy)
149 # PDF
150 if "Treść / Zawartość" in sections_dict:
151 pdf_a_tag = sections_dict["Treść / Zawartość"].select_one("a")
152 if not pdf_a_tag:
153 raise ValueError("Cannot find pdf for article")
154 pdf_url = pdf_a_tag.get("href")
155 if not isinstance(pdf_url, str):
156 raise ValueError("Cannot parse pdf url for article")
157 if not pdf_url.startswith("http"):
158 pdf_url = self.source_website + "/" + pdf_url
159 add_pdf_link_to_xarticle(xarticle, pdf_url)
160 else:
161 self.logger.info("PDF not found", extra={"pid": xarticle.pid})
163 # Lang
164 xarticle.lang = cleanup_str(sections_dict["Języki publikacji"].text.lower())
165 if len(xarticle.lang) > 3:
166 if xarticle.lang == "pl fr":
167 xarticle.lang = "pl"
168 self.logger.info(
169 f"[{xarticle.pid}] Patch : set article lang to 'pl' (was 'pl fr' before)",
170 extra={"pid": xarticle.pid},
171 )
172 else:
173 raise ValueError("Cannot parse article lang")
175 # Abstract
176 if "Abstrakty" in sections_dict:
177 abstract_divs = sections_dict["Abstrakty"].select("div.listing-row")
178 for div in abstract_divs:
179 lang = "und"
180 lang_div = div.select_one("div.articleDetails-langCell")
181 if lang_div:
182 lang = cleanup_str(lang_div.text).lower()
183 text_div = div.select_one("div.articleDetails-abstract")
184 if not text_div:
185 raise ValueError(
186 "Error while parsing abstract : abstract presence detected, but abstract cannot be parsed"
187 )
188 abstract_text = cleanup_str(text_div.text)
189 if abstract_text != "-":
190 xarticle.abstracts.append(create_abstract(value_tex=abstract_text, lang=lang))
192 # Keywords
193 if "Słowa kluczowe" in sections_dict:
194 keywords_lists = sections_dict["Słowa kluczowe"].select("div.listing-row")
195 for list in keywords_lists:
196 lang = "und"
197 lang_div = list.select_one("div.articleDetails-langCell")
198 keywords_a_tags = list.select("a")
199 for a_tag in keywords_a_tags:
200 subject = create_subj()
201 subject["value"] = a_tag.text
202 subject["lang"] = lang
203 xarticle.kwds.append(subject)
204 # Page
205 if "Strony" in sections_dict:
206 set_pages(xarticle, cleanup_str(sections_dict["Strony"].text))
208 return xarticle
210 def parse_author(self, content: str):
211 author = create_contributor()
212 sections_dict = self.parse_dmlpl_generic_page(content)
213 author["last_name"] = cleanup_str(sections_dict["Nazwisko"].text)
214 author["first_name"] = cleanup_str(sections_dict["Imię"].text)
215 if len(author["last_name"]) == 0 or len(author["first_name"]) == 0:
216 author["string_name"] = cleanup_str(sections_dict["Twórca"].text)
217 return author