Coverage for src/crawler/utils.py: 40%

144 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-09-23 14:47 +0000

1import base64 

2import hashlib 

3import json 

4import logging 

5import os 

6import pickle 

7import re 

8import unicodedata 

9from collections.abc import Callable 

10from datetime import datetime, timezone 

11from functools import lru_cache, wraps 

12from typing import Literal, TypeVar 

13 

14import regex 

15from bs4 import BeautifulSoup 

16from django.conf import settings 

17from django.contrib.auth.models import User 

18from history.model_data import HistoryEventDict, HistoryEventType 

19from history.models import HistoryEventStatus 

20 

21# from ptf.models import ResourceId 

22from history.utils import insert_history_event 

23from ptf import model_helpers 

24from ptf.cmds import ptf_cmds 

25from ptf.cmds.xml.xml_utils import escape 

26from ptf.exceptions import ResourceDoesNotExist 

27from ptf.model_data import ResourceData, create_extlink, create_publicationdata 

28from ptf.models import Collection 

29from pymongo import MongoClient 

30from pymongo.synchronous.database import Database 

31 

32from crawler.types import JSONCol 

33 

34logger = logging.getLogger(__name__) 

35 

36 

37def insert_crawl_event_in_history( 

38 colid: str, 

39 source_domain: str, 

40 username: str, 

41 status: HistoryEventStatus, 

42 tasks_count, 

43 message: str, 

44 event_type: HistoryEventType = "import-collection", 

45 title=None, 

46): 

47 collection = model_helpers.get_collection(colid, sites=False) 

48 user = User.objects.get(username=username) 

49 

50 event_data: HistoryEventDict = { 

51 "type": event_type, 

52 "pid": f"{colid}-{source_domain}", 

53 "col": collection, 

54 "source": source_domain, 

55 "status": status, 

56 "title": collection.title_html if collection else (title or colid), 

57 "userid": user.pk, 

58 "type_error": "", 

59 "message": message, 

60 } 

61 

62 insert_history_event(event_data) 

63 

64 

65def col_has_source(col: JSONCol, filter: str): 

66 return any(source for source in col["sources"] if source == filter) 

67 

68 

69def get_cols_by_source(source: str) -> list[JSONCol]: 

70 """ 

71 Get all cols by source 

72 @param source: str 

73 @return: list of collections 

74 """ 

75 data = get_all_cols() 

76 

77 return [col for col in data.values() if col_has_source(col, source)] 

78 

79 

80def get_all_cols_by_source(): 

81 """ 

82 Get all cols by source 

83 @return: dict of collections by source 

84 """ 

85 data = get_all_cols() 

86 

87 sources: dict[str, list] = {} 

88 for col in data.values(): 

89 for source in col["sources"]: 

90 if source not in sources: 

91 sources[source] = [] 

92 sources[source].append(col) 

93 

94 return sources 

95 

96 

97@lru_cache(maxsize=None) 

98def get_all_cols() -> dict[str, JSONCol]: 

99 with open( 

100 os.path.dirname(os.path.abspath(__file__)) + "/data/all_cols.json", encoding="utf8" 

101 ) as data_collections: 

102 return json.load(data_collections) 

103 

104 

105def get_or_create_collection(pid: str): 

106 """ 

107 Creates a Collection based on its pid. 

108 The pid has to be in the list of collections given by the Documentation team (CSV then transformed in JSON) 

109 """ 

110 

111 all_collections = get_all_cols() 

112 

113 if pid not in all_collections: 

114 raise ValueError(f"{pid} is not listed in all_cols.csv") 

115 

116 col_data = [item for item in all_collections.items() if item[0] == pid][0][1] 

117 

118 collection: Collection | None = model_helpers.get_collection(pid, sites=False) 

119 

120 if not collection: 

121 p = model_helpers.get_provider("mathdoc-id") 

122 

123 xcol = create_publicationdata() 

124 xcol.coltype = col_data["type"] 

125 xcol.pid = pid 

126 xcol.title_tex = col_data["title"] 

127 # Mis en commentaire car trop tôt, Taban n'a pas encore validé les ISSNs 

128 # xcol.e_issn = col_data["ISSN_électronique"] 

129 # xcol.issn = col_data["ISSN_papier"] 

130 xcol.title_html = col_data["title"] 

131 xcol.title_xml = f"<title-group><title>{col_data['title']}</title></title-group>" 

132 xcol.lang = "en" 

133 

134 cmd = ptf_cmds.addCollectionPtfCmd({"xobj": xcol}) 

135 cmd.set_provider(p) 

136 collection = cmd.do() 

137 

138 # Mis en commentaire car trop tôt, Taban n'a pas encore validé les ISSNs 

139 # if col_data["ISSN_électronique"] != "": 

140 # e_issn = { 

141 # "resource_id": collection.resource_ptr_id, 

142 # "id_type": "e_issn", 

143 # "id_value": col_data["ISSN_électronique"], 

144 # } 

145 # ResourceId.objects.create(**e_issn) 

146 # 

147 # if col_data["ISSN_papier"] != "": 

148 # issn = { 

149 # "resource_id": collection.resource_ptr_id, 

150 # "id_type": "issn", 

151 # "id_value": col_data["ISSN_papier"], 

152 # } 

153 # ResourceId.objects.create(**issn) 

154 

155 if not collection: 

156 raise ResourceDoesNotExist(f"Resource {pid} does not exist") 

157 

158 return collection 

159 

160 

161def cleanup_str(input: str, unsafe=False): 

162 # some white spaces aren't actual space characters, like \xa0 

163 input = unicodedata.normalize("NFKC", input) 

164 # 

165 input = re.sub(r"[\x7f]+", "", input) 

166 # remove useless continuous \n and spaces from the string 

167 input = re.sub(r"[\n\t\r ]+", " ", input).strip() 

168 if unsafe: 168 ↛ 169line 168 didn't jump to line 169 because the condition on line 168 was never true

169 return input 

170 return escape(input) 

171 

172 

173def add_pdf_link_to_xarticle( 

174 xarticle: ResourceData, 

175 pdf_url: str, 

176 mimetype: Literal["application/pdf", "application/x-tex", "text/html"] = "application/pdf", 

177): 

178 xarticle.streams.append( 

179 { 

180 "rel": "full-text", 

181 "mimetype": mimetype, 

182 "location": pdf_url, 

183 "base": "", 

184 "text": "Full Text", 

185 } 

186 ) 

187 

188 # The pdf url is already added as a stream (just above) but might be replaced by a file later on. 

189 # Keep the pdf url as an Extlink if we want to propose both option: 

190 # - direct download of a local PDF 

191 # - URL to the remote PDF 

192 if mimetype == "application/pdf": 192 ↛ 194line 192 didn't jump to line 194 because the condition on line 192 was always true

193 rel = "article-pdf" 

194 elif mimetype == "text/html": 

195 rel = "article-html" 

196 else: 

197 rel = "article-tex" 

198 ext_link = create_extlink(rel=rel, location=pdf_url) 

199 xarticle.ext_links.append(ext_link) 

200 

201 

202def add_source_link_to_xarticle(xarticle, url: str, domain: str): 

203 ext_link = create_extlink() 

204 ext_link["rel"] = "source" 

205 ext_link["location"] = url 

206 ext_link["metadata"] = domain 

207 xarticle.ext_links.append(ext_link) 

208 

209 

210def regex_to_dict(pattern: str, value: str, *, error_msg="Regex failed to parse"): 

211 issue_search = regex.search(pattern, value) 

212 if not issue_search: 212 ↛ 213line 212 didn't jump to line 213 because the condition on line 212 was never true

213 raise ValueError(error_msg) 

214 

215 return issue_search.groupdict() 

216 

217 

218def get_base(soup: BeautifulSoup, default: str): 

219 base_tag = soup.select_one("head base") 

220 if not base_tag: 

221 return default 

222 base = base_tag.get("href") 

223 if not isinstance(base, str): 

224 raise ValueError("Cannot parse base href") 

225 return base 

226 

227 

228try: 

229 from crawler.tests.data_generation.decorators import skip_generation 

230except ImportError: 

231 

232 def skip_generation(func): 

233 def wrapper(*args, **kwargs): 

234 return func(*args, **kwargs) 

235 

236 return wrapper 

237 

238 

239class PickleSerializer: 

240 def serialize(self, obj): 

241 return base64.b64encode(pickle.dumps(obj, protocol=-1)) 

242 

243 def deserialize(self, serialized): 

244 return pickle.loads(base64.b64decode(serialized)) 

245 

246 

247RT = TypeVar("RT") 

248 

249 

250def mongo_cache( 

251 db_conn: Database = MongoClient(host=getattr(settings, "MONGO_HOSTNAME", "localhost"))[ 

252 "crawler_func_cache" 

253 ], 

254 prefix="cache_", 

255 capped=True, 

256 capped_size=1000000000, 

257 hash_keys=True, 

258 serializer=PickleSerializer, 

259) -> Callable[[Callable[..., RT]], Callable[..., RT]]: 

260 """Helper decorator to speedup local development 

261 ```py 

262 

263 

264 @mongo_cache( 

265 db_conn=MongoClient(host=getattr(settings, "MONGO_HOSTNAME", "localhost"))[ 

266 "crawler_func_cache" 

267 ] 

268 ) 

269 def parse_collection_content(self, content): 

270 ... 

271 ``` 

272 """ 

273 

274 def decorator(func: Callable[..., RT]) -> Callable[..., RT]: 

275 serializer_ins = serializer() 

276 col_name = f"{prefix}{func.__name__}" 

277 if capped: 

278 db_conn.create_collection(col_name, capped=capped, size=capped_size) 

279 

280 cache_col = db_conn[col_name] 

281 cache_col.create_index("key", unique=True) 

282 cache_col.create_index("date", expireAfterSeconds=86400) 

283 

284 @wraps(func) 

285 def wrapped_func(*args, **kwargs) -> RT: 

286 cache_key = pickle.dumps((args[1:], kwargs), protocol=-1) 

287 if hash_keys: 

288 cache_key = hashlib.md5(cache_key).hexdigest() 

289 else: 

290 cache_key = base64.b64encode(cache_key) 

291 

292 cached_obj = cache_col.find_one(dict(key=cache_key)) 

293 if cached_obj: 

294 return serializer_ins.deserialize(cached_obj["result"]) 

295 

296 ret = func(*args, **kwargs) 

297 cache_col.update_one( 

298 {"key": cache_key}, 

299 { 

300 "$set": { 

301 "result": serializer_ins.serialize(ret), 

302 "date": datetime.now(timezone.utc), 

303 } 

304 }, 

305 upsert=True, 

306 ) 

307 

308 return ret 

309 

310 return wrapped_func 

311 

312 return decorator