lms.util.net
Utilities for network and HTTP.
1""" 2Utilities for network and HTTP. 3""" 4 5import re 6import typing 7import urllib.parse 8 9import bs4 10import edq.net.exchange 11import edq.util.hash 12import edq.util.json 13import requests 14 15import lms.model.constants 16 17STANDARDIZED_TIMESTAMP: str = '123456789' 18STANDARDIZED_SESSION_KEY: str = 'abcABC123' 19STANDARDIZED_RANDOM_STRING: str = 'abc123' 20 21CANVAS_CLEAN_REMOVE_CONTENT_KEYS: typing.List[str] = [ 22 'created_at', 23 'ics', 24 'last_activity_at', 25 'lti_context_id', 26 'preview_url', 27 'secure_params', 28 'total_activity_time', 29 'updated_at', 30 'url', 31 'uuid', 32] 33""" Keys to remove from Canvas content. """ 34 35BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS: typing.List[str] = [ 36 'created', 37 'modified', 38] 39""" Keys to remove from Blackboard content. """ 40 41BLACKBOARD_CLEAN_REMOVE_HEADERS: typing.Set[str] = { 42 'access-control-allow-origin', 43 'content-encoding', 44 'content-language', 45 'expires', 46 'last-modified', 47 'p3p', 48 'strict-transport-security', 49 'transfer-encoding', 50 'vary', 51 'x-blackboard-xsrf', 52} 53""" Keys to remove from Blackboard headers. """ 54 55MOODLE_CLEAN_REMOVE_HEADERS: typing.Set[str] = { 56 'accept-ranges', 57 'content-encoding', 58 'content-language', 59 'content-script-type', 60 'content-style-type', 61 'expires', 62 'keep-alive', 63 'last-modified', 64 'transfer-encoding', 65 'vary', 66} 67""" Keys to remove from Moodle headers. """ 68 69MOODLE_FINALIZE_REMOVE_PARAMS: typing.Set[str] = { 70 'logintoken', 71} 72""" Keys to remove from Moodle headers. """ 73 74MOODLE_HTML_CLEAN: typing.Dict[str, typing.Dict[str, typing.Any]] = { 75 r'/login/index\.php': { 76 'delete_elements': ['script', 'footer'], 77 'remove_all_attrs': ['body', 'div'], 78 'hoist_elements': { 79 ('div#page-wrapper', 'input[name="logintoken"]'), 80 }, 81 }, 82 r'/user/index\.php\?id=(\d+)': { 83 'delete_elements': [ 84 'tr.emptyrow', 85 'div[data-status="Active"]', 86 'i', 87 'th.a', 88 ], 89 'attrs_to_keep': { 90 'a': ['data-column'], 91 'th': ['class'], 92 'td': ['class'], 93 }, 94 'remove_all_attrs': ['tr'], 95 'final_selectors': ['table#participants'], 96 }, 97 r'/user/profile.php': { 98 'filter_elements_by_descendant': { 99 'div.card-body': ('h3', 'Course details'), 100 }, 101 'hoist_elements': { 102 ('div.card-body ul', 'div.card-body ul li ul'), 103 }, 104 'final_selectors': ['div.card-body'], 105 }, 106 r'/grade/report/grader/index\.php\?id=(\d+)': { 107 'delete_elements': [ 108 'caption', 109 'tr[class=""]', 110 'tr[class="avg"]', 111 'i', 112 'button', 113 'a.dropdown-item', 114 'img', 115 'label', 116 ], 117 'attrs_to_keep': { 118 'input': ['max', 'name', 'data-context'], 119 'a': ['class'], 120 'th': ['class', 'data-itemid'], 121 'td': ['class'], 122 'td input': ['max'], 123 }, 124 'classes_to_keep': { 125 'th': ['item', re.compile(r'c\d+')], 126 'td': [re.compile(r'c\d+')], 127 }, 128 'remove_all_attrs': [ 129 'div', 130 'tr', 131 ], 132 'hoist_elements': { 133 ('div.d-flex.flex-column.h-100', 'a.gradeitemheader'), 134 ('div.d-flex.flex-column.h-100', 'input[title="Grade"]'), 135 }, 136 'replacements': [ 137 (r'\n', ''), 138 (r'\s+aria-(?:label|hidden|expanded)="[^"]*"', ''), 139 (r'\s+role\s*=\s*(?:"[^"]*"|\'[^\']*\')', ''), 140 (r'<div>', ''), 141 (r'</div>', ''), 142 (r'(<script>)[\s\S]*?(</script>)', rf'\1M.cfg = {{"sesskey":"{STANDARDIZED_SESSION_KEY}"}}\2') 143 ], 144 'final_selectors': ['head script:not([type])', 'table#user-grades', 'input[name=setmode]'], 145 }, 146} 147""" A mapping of Moodle URL patterns to clean_html() kwargs. """ 148 149def clean_lms_response(response: requests.Response, body: str) -> str: 150 """ 151 A ResponseModifierFunction that attempt to identify 152 if the requests comes from a Learning Management System (LMS), 153 and clean the response accordingly. 154 """ 155 156 # Check the standard LMS Toolkit backend header. 157 raw_backend_type = response.headers.get(lms.model.constants.HEADER_KEY_BACKEND, '').lower() 158 159 if (raw_backend_type == lms.model.constants.BackendType.CANVAS.value): 160 return clean_canvas_response(response, body) 161 162 if (raw_backend_type == lms.model.constants.BackendType.MOODLE.value): 163 return clean_moodle_response(response, body) 164 165 # Try looking inside the header keys. 166 for key in response.headers: 167 key = key.lower().strip() 168 169 if ('blackboard' in key): 170 return clean_blackboard_response(response, body) 171 172 if ('canvas' in key): 173 return clean_canvas_response(response, body) 174 175 if ('moodle' in key): 176 return clean_moodle_response(response, body) 177 178 return body 179 180def clean_blackboard_response(response: requests.Response, body: str) -> str: 181 """ 182 See clean_lms_response(), but specifically for the Blackboard LMS. 183 This function will: 184 - Call _clean_base_response(). 185 - Remove specific headers. 186 """ 187 188 body = _clean_base_response(response, body) 189 190 # Work on both request and response headers. 191 remove_headers(response, BLACKBOARD_CLEAN_REMOVE_HEADERS) 192 193 # Most blackboard responses are JSON. 194 try: 195 data = edq.util.json.loads(body, strict = True) 196 except Exception: 197 # Response is not JSON. 198 return body 199 200 # Remove any content keys. 201 _recursive_remove_keys(data, set(BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS)) 202 203 # Convert body back to a string. 204 body = edq.util.json.dumps(data) 205 206 return body 207 208def clean_canvas_response(response: requests.Response, body: str) -> str: 209 """ 210 See clean_lms_response(), but specifically for the Canvas LMS. 211 This function will: 212 - Call _clean_base_response(). 213 - Remove content keys: [last_activity_at, total_activity_time] 214 """ 215 216 body = _clean_base_response(response, body) 217 url = str(response.request.url) 218 219 if ('/files_api' in url): 220 # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing. 221 parts = urllib.parse.urlsplit(response.headers['location']) 222 parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '') 223 response.headers['location'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG) 224 225 # Most canvas responses are JSON. 226 try: 227 data = edq.util.json.loads(body, strict = True) 228 except Exception: 229 # Response is not JSON. 230 return body 231 232 # Remove any content keys. 233 _recursive_remove_keys(data, set(CANVAS_CLEAN_REMOVE_CONTENT_KEYS)) 234 235 # Handle endpoint-specific cases. 236 if ('submissions/update_grades' in url): 237 data.pop('id', None) 238 elif (re.search(r'api/v1/courses/\w+/files', url) is not None): 239 # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing. 240 parts = urllib.parse.urlsplit(data['upload_url']) 241 parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '') 242 data['upload_url'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG) 243 244 # Convert body back to a string. 245 body = edq.util.json.dumps(data) 246 247 return body 248 249def finalize_canvas_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange: 250 """ Finalize Canvas exchanges. """ 251 252 if (re.search(r'^api/v1/courses/\w+/files$', exchange.url_path) is not None): 253 # Clean out random data for the file upload. 254 data = edq.util.json.loads(str(exchange.response_body)) 255 256 body_data = { 257 'upload_params': { 258 'Filename': data['upload_params']['Filename'], 259 }, 260 'upload_url': data['upload_url'], 261 } 262 263 exchange.response_body = edq.util.json.dumps(body_data) 264 elif (exchange.url_path == 'files_api'): 265 # File upload calls contain several random pieces of information generated from the server. 266 # Drop information we don't need in testing and make the rest consistent. 267 268 filename = exchange.parameters['Filename'] 269 exchange.parameters = {'filename': filename} 270 exchange.files = [] 271 272 location = exchange.response_headers.get('location', None) 273 if (location is not None): 274 location = re.sub(r'uuid=.*$', f"uuid={edq.util.hash.sha256_hex(filename)}", str(location)) 275 276 exchange.response_headers['location'] = location 277 elif (re.search(r'^api/v1/files/\w+/create_success$', exchange.url_path) is not None): 278 # Change the uuid parameter to match the one set in files_api (which redirects to this URL). 279 data = edq.util.json.loads(str(exchange.response_body)) 280 exchange.parameters['uuid'] = edq.util.hash.sha256_hex(data['display_name']) 281 exchange.response_body = edq.util.json.dumps({'id': data['id']}) 282 283 return exchange 284 285def clean_moodle_response(response: requests.Response, body: str) -> str: 286 """ 287 See clean_lms_response(), but specifically for the Moodle LMS. 288 This function will: 289 - Call _clean_base_response(). 290 """ 291 292 body = _clean_base_response(response, body) 293 294 # Standardize timestamp. 295 current_timestamp_match = re.search(r"boost/theme/(\d{10})/favicon", body) 296 if (current_timestamp_match is not None): 297 body = body.replace(current_timestamp_match.group(1), STANDARDIZED_TIMESTAMP) 298 299 # Standardize session key. 300 session_key_match = re.search(r'"sesskey":"([^"]+)"', body) 301 if (session_key_match is not None): 302 body = body.replace(session_key_match.group(1), STANDARDIZED_SESSION_KEY) 303 304 # Standardize "random" string. 305 random_string_match = re.search(r"'random([a-z0-9]+)'", body) 306 if (random_string_match is not None): 307 body = body.replace(random_string_match.group(1), STANDARDIZED_RANDOM_STRING) 308 309 # Standardize logintoken. 310 logintoken_match = re.search(r'name="logintoken" value="(\w+)"', body) 311 if (logintoken_match is not None): 312 body = body.replace(logintoken_match.group(1), STANDARDIZED_SESSION_KEY) 313 314 # Standardize last access to course. 315 last_access_match = re.search(r'(\d+) secs', body) 316 if (last_access_match is not None): 317 body = body.replace(last_access_match.group(0), f'{STANDARDIZED_TIMESTAMP} secs') 318 319 # Work on both request and response headers. 320 remove_headers(response, MOODLE_CLEAN_REMOVE_HEADERS) 321 322 # Clean HTML responses. 323 for (pattern, clean_params) in MOODLE_HTML_CLEAN.items(): 324 if (re.search(pattern, response.url.strip())): 325 body = clean_html(body, **clean_params) 326 327 return body 328 329def clean_html( 330 html: str, 331 filter_elements_by_descendant: typing.Union[typing.Dict[str, typing.Tuple[str, str]], None] = None, 332 delete_elements: typing.Union[typing.List[str], None] = None, 333 attrs_to_keep: typing.Union[typing.Dict[str, typing.List[str]], None] = None, 334 classes_to_keep: typing.Union[typing.Dict[str, typing.List[typing.Union[str, re.Pattern]]], None] = None, 335 remove_all_attrs: typing.Union[typing.List[str], None] = None, 336 hoist_elements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None, 337 replacements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None, 338 final_selectors: typing.Union[typing.List[str], None] = None, 339 ) -> str: 340 """ 341 General purpose HTML cleaning function. 342 343 filter_elements_by_descendant: { selector: (descendant_selector, text), ... } 344 Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text. 345 For example: 346 ``` 347 >>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>' 348 >>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')}) 349 <div><h2>Hello World!</h2></div> 350 ``` 351 352 delete_elements: [ selector, ... ] 353 Deletes any matching elements. 354 355 attrs_to_keep: { selector: [attribute, ...], ... } 356 Keeps only the listed attributes of elements matching the selector. 357 358 classes_to_keep: { selector: [class/pattern, ...], ... } 359 Keeps only listed classes, or classes matching the pattern, of elements matching the selector. 360 361 remove_all_attrs: [ selector, ... ] 362 Removes all attributes of matching elements. 363 364 hoist_elements: [ (parent_selector, child_selector), ... ] 365 For each pair of selectors, the parent element is replaced by the first matching child within the parent. 366 No replacement occurs if both matches are not found. 367 368 replacements: [ (pattern, replacement), ... ] 369 Performs a replacement for all regex matches. 370 371 final_selectors: [ selector, ... ] 372 Replaces html with all matched elements. 373 Useful for targeting only the necessary parts of the response. 374 Defaults to ['body'] if None. 375 """ 376 377 if (filter_elements_by_descendant is None): 378 filter_elements_by_descendant = {} 379 380 if (delete_elements is None): 381 delete_elements = [] 382 383 if (attrs_to_keep is None): 384 attrs_to_keep = {} 385 386 if (classes_to_keep is None): 387 classes_to_keep = {} 388 389 if (remove_all_attrs is None): 390 remove_all_attrs = [] 391 392 if (hoist_elements is None): 393 hoist_elements = [] 394 395 if (replacements is None): 396 replacements = [(r'\n', '')] 397 398 if (final_selectors is None): 399 final_selectors = ['body'] 400 401 document = bs4.BeautifulSoup(html, 'html.parser') 402 403 # Filter Elements by Descendant 404 for (selector, (descendant_selector, text)) in filter_elements_by_descendant.items(): 405 for element in document.select(selector): 406 descendant = element.select_one(descendant_selector) 407 if ((descendant is None) or (descendant.get_text() != text)): 408 element.decompose() 409 410 # Delete Elements 411 for selector in delete_elements: 412 for element in document.select(selector): 413 element.decompose() 414 415 # Keep Attributes 416 for (element_selector, attrs) in attrs_to_keep.items(): 417 for element in document.select(element_selector): 418 # Remove extra attributes by keeping only select attributes and replacing the existing attribute dict. 419 element.attrs = {attr: element.attrs[attr] for attr in attrs if (attr in element.attrs)} 420 421 # Keep Classes 422 for (element_selector, keep_classes) in classes_to_keep.items(): 423 for element in document.select(element_selector): 424 classes = element.get('class') 425 if ((classes is None) or (len(classes) == 0)): 426 continue 427 428 # Keep only classes listed for this selector. 429 kept = [] 430 for keep_class in keep_classes: 431 for class_name in classes: 432 if (((isinstance(keep_class, re.Pattern)) and (keep_class.fullmatch(class_name))) or (keep_class == class_name)): 433 kept.append(class_name) 434 435 element['class'] = kept # type: ignore[assignment] 436 437 # Remove All Attributes 438 for selector in remove_all_attrs: 439 for element in document.select(selector): 440 element.attrs.clear() 441 442 # Element Hoisting 443 for (parent_selector, child_selector) in hoist_elements: 444 for parent in document.select(parent_selector): 445 child = parent.select_one(child_selector) 446 if (child is None): 447 continue 448 449 parent.replace_with(child.extract()) 450 451 # Final Selectors 452 elements = [] 453 for final_selector in final_selectors: 454 for element in document.select(final_selector): 455 elements.append(element) 456 457 document_string = "".join([str(element) for element in elements]) 458 459 # Replacements 460 for (pattern, replacement) in replacements: 461 document_string = re.sub(pattern, replacement, document_string) 462 463 return document_string 464 465def remove_headers(response: requests.Response, headers_to_remove: typing.Set[str]) -> None: 466 """ 467 Remove headers from response and response's request. 468 """ 469 470 for headers in [response.headers, response.request.headers]: 471 for key in list(headers.keys()): # type: ignore[attr-defined] 472 if (key.strip().lower() in headers_to_remove): 473 headers.pop(key, None) # type: ignore[attr-defined] 474 475def finalize_moodle_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange: 476 """ Finalize Moodle exchanges. """ 477 478 for param in MOODLE_FINALIZE_REMOVE_PARAMS: 479 exchange.parameters.pop(param, None) 480 481 # Standardize session key. 482 if ('sesskey' in exchange.parameters): 483 exchange.parameters['sesskey'] = STANDARDIZED_SESSION_KEY 484 485 return exchange 486 487def _clean_base_response(response: requests.Response, body: str, 488 keep_headers: typing.Union[typing.List[str], None] = None) -> str: 489 """ 490 Do response cleaning that is common amongst all backend types. 491 This function will: 492 - Remove X- headers. 493 """ 494 495 # Index requests are generally for identification, and we use headers. 496 path = urllib.parse.urlparse(response.request.url).path.strip() 497 if (path in ['', '/']): 498 body = '' 499 500 for key in list(response.headers.keys()): 501 key = key.strip().lower() 502 if ((keep_headers is not None) and (key in keep_headers)): 503 continue 504 505 if (key.startswith('x-')): 506 response.headers.pop(key, None) 507 508 return body 509 510def _recursive_remove_keys(data: typing.Any, remove_keys: typing.Set[str]) -> None: 511 """ 512 Recursively descend through the given and remove any instance to the given key from any dictionaries. 513 The data should only be simple types (POD, dicts, lists, tuples). 514 """ 515 516 if (isinstance(data, (list, tuple))): 517 for item in data: 518 _recursive_remove_keys(item, remove_keys) 519 elif (isinstance(data, dict)): 520 for key in list(data.keys()): 521 if (key in remove_keys): 522 del data[key] 523 else: 524 _recursive_remove_keys(data[key], remove_keys) 525 526def parse_cookies( 527 text_cookies: typing.Union[str, None], 528 strip_key_prefix: bool = True, 529 ) -> typing.Dict[str, typing.Any]: 530 """ Parse cookies out of a text string. """ 531 532 cookies: typing.Dict[str, typing.Any] = {} 533 534 if (text_cookies is None): 535 return cookies 536 537 text_cookies = text_cookies.strip() 538 if (len(text_cookies) == 0): 539 return cookies 540 541 for cookie in text_cookies.split('; '): 542 parts = cookie.split('=', maxsplit = 1) 543 544 key = parts[0].lower() 545 546 if (strip_key_prefix): 547 key = key.split(', ')[-1] 548 549 if (len(parts) == 1): 550 cookies[key] = True 551 else: 552 cookies[key] = parts[1] 553 554 return cookies
Keys to remove from Canvas content.
Keys to remove from Blackboard content.
Keys to remove from Blackboard headers.
Keys to remove from Moodle headers.
Keys to remove from Moodle headers.
A mapping of Moodle URL patterns to clean_html() kwargs.
150def clean_lms_response(response: requests.Response, body: str) -> str: 151 """ 152 A ResponseModifierFunction that attempt to identify 153 if the requests comes from a Learning Management System (LMS), 154 and clean the response accordingly. 155 """ 156 157 # Check the standard LMS Toolkit backend header. 158 raw_backend_type = response.headers.get(lms.model.constants.HEADER_KEY_BACKEND, '').lower() 159 160 if (raw_backend_type == lms.model.constants.BackendType.CANVAS.value): 161 return clean_canvas_response(response, body) 162 163 if (raw_backend_type == lms.model.constants.BackendType.MOODLE.value): 164 return clean_moodle_response(response, body) 165 166 # Try looking inside the header keys. 167 for key in response.headers: 168 key = key.lower().strip() 169 170 if ('blackboard' in key): 171 return clean_blackboard_response(response, body) 172 173 if ('canvas' in key): 174 return clean_canvas_response(response, body) 175 176 if ('moodle' in key): 177 return clean_moodle_response(response, body) 178 179 return body
A ResponseModifierFunction that attempt to identify if the requests comes from a Learning Management System (LMS), and clean the response accordingly.
181def clean_blackboard_response(response: requests.Response, body: str) -> str: 182 """ 183 See clean_lms_response(), but specifically for the Blackboard LMS. 184 This function will: 185 - Call _clean_base_response(). 186 - Remove specific headers. 187 """ 188 189 body = _clean_base_response(response, body) 190 191 # Work on both request and response headers. 192 remove_headers(response, BLACKBOARD_CLEAN_REMOVE_HEADERS) 193 194 # Most blackboard responses are JSON. 195 try: 196 data = edq.util.json.loads(body, strict = True) 197 except Exception: 198 # Response is not JSON. 199 return body 200 201 # Remove any content keys. 202 _recursive_remove_keys(data, set(BLACKBOARD_CLEAN_REMOVE_CONTENT_KEYS)) 203 204 # Convert body back to a string. 205 body = edq.util.json.dumps(data) 206 207 return body
See clean_lms_response(), but specifically for the Blackboard LMS. This function will:
- Call _clean_base_response().
- Remove specific headers.
209def clean_canvas_response(response: requests.Response, body: str) -> str: 210 """ 211 See clean_lms_response(), but specifically for the Canvas LMS. 212 This function will: 213 - Call _clean_base_response(). 214 - Remove content keys: [last_activity_at, total_activity_time] 215 """ 216 217 body = _clean_base_response(response, body) 218 url = str(response.request.url) 219 220 if ('/files_api' in url): 221 # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing. 222 parts = urllib.parse.urlsplit(response.headers['location']) 223 parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '') 224 response.headers['location'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG) 225 226 # Most canvas responses are JSON. 227 try: 228 data = edq.util.json.loads(body, strict = True) 229 except Exception: 230 # Response is not JSON. 231 return body 232 233 # Remove any content keys. 234 _recursive_remove_keys(data, set(CANVAS_CLEAN_REMOVE_CONTENT_KEYS)) 235 236 # Handle endpoint-specific cases. 237 if ('submissions/update_grades' in url): 238 data.pop('id', None) 239 elif (re.search(r'api/v1/courses/\w+/files', url) is not None): 240 # Replace the scheme + netloc (host + port) with a slug that can be replaced when testing. 241 parts = urllib.parse.urlsplit(data['upload_url']) 242 parts = parts._replace(netloc = lms.model.constants.SERVER_SLUG, scheme = '') 243 data['upload_url'] = parts.geturl().replace(f"//{lms.model.constants.SERVER_SLUG}", lms.model.constants.SERVER_SLUG) 244 245 # Convert body back to a string. 246 body = edq.util.json.dumps(data) 247 248 return body
See clean_lms_response(), but specifically for the Canvas LMS. This function will:
- Call _clean_base_response().
- Remove content keys: [last_activity_at, total_activity_time]
250def finalize_canvas_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange: 251 """ Finalize Canvas exchanges. """ 252 253 if (re.search(r'^api/v1/courses/\w+/files$', exchange.url_path) is not None): 254 # Clean out random data for the file upload. 255 data = edq.util.json.loads(str(exchange.response_body)) 256 257 body_data = { 258 'upload_params': { 259 'Filename': data['upload_params']['Filename'], 260 }, 261 'upload_url': data['upload_url'], 262 } 263 264 exchange.response_body = edq.util.json.dumps(body_data) 265 elif (exchange.url_path == 'files_api'): 266 # File upload calls contain several random pieces of information generated from the server. 267 # Drop information we don't need in testing and make the rest consistent. 268 269 filename = exchange.parameters['Filename'] 270 exchange.parameters = {'filename': filename} 271 exchange.files = [] 272 273 location = exchange.response_headers.get('location', None) 274 if (location is not None): 275 location = re.sub(r'uuid=.*$', f"uuid={edq.util.hash.sha256_hex(filename)}", str(location)) 276 277 exchange.response_headers['location'] = location 278 elif (re.search(r'^api/v1/files/\w+/create_success$', exchange.url_path) is not None): 279 # Change the uuid parameter to match the one set in files_api (which redirects to this URL). 280 data = edq.util.json.loads(str(exchange.response_body)) 281 exchange.parameters['uuid'] = edq.util.hash.sha256_hex(data['display_name']) 282 exchange.response_body = edq.util.json.dumps({'id': data['id']}) 283 284 return exchange
Finalize Canvas exchanges.
286def clean_moodle_response(response: requests.Response, body: str) -> str: 287 """ 288 See clean_lms_response(), but specifically for the Moodle LMS. 289 This function will: 290 - Call _clean_base_response(). 291 """ 292 293 body = _clean_base_response(response, body) 294 295 # Standardize timestamp. 296 current_timestamp_match = re.search(r"boost/theme/(\d{10})/favicon", body) 297 if (current_timestamp_match is not None): 298 body = body.replace(current_timestamp_match.group(1), STANDARDIZED_TIMESTAMP) 299 300 # Standardize session key. 301 session_key_match = re.search(r'"sesskey":"([^"]+)"', body) 302 if (session_key_match is not None): 303 body = body.replace(session_key_match.group(1), STANDARDIZED_SESSION_KEY) 304 305 # Standardize "random" string. 306 random_string_match = re.search(r"'random([a-z0-9]+)'", body) 307 if (random_string_match is not None): 308 body = body.replace(random_string_match.group(1), STANDARDIZED_RANDOM_STRING) 309 310 # Standardize logintoken. 311 logintoken_match = re.search(r'name="logintoken" value="(\w+)"', body) 312 if (logintoken_match is not None): 313 body = body.replace(logintoken_match.group(1), STANDARDIZED_SESSION_KEY) 314 315 # Standardize last access to course. 316 last_access_match = re.search(r'(\d+) secs', body) 317 if (last_access_match is not None): 318 body = body.replace(last_access_match.group(0), f'{STANDARDIZED_TIMESTAMP} secs') 319 320 # Work on both request and response headers. 321 remove_headers(response, MOODLE_CLEAN_REMOVE_HEADERS) 322 323 # Clean HTML responses. 324 for (pattern, clean_params) in MOODLE_HTML_CLEAN.items(): 325 if (re.search(pattern, response.url.strip())): 326 body = clean_html(body, **clean_params) 327 328 return body
See clean_lms_response(), but specifically for the Moodle LMS. This function will:
- Call _clean_base_response().
330def clean_html( 331 html: str, 332 filter_elements_by_descendant: typing.Union[typing.Dict[str, typing.Tuple[str, str]], None] = None, 333 delete_elements: typing.Union[typing.List[str], None] = None, 334 attrs_to_keep: typing.Union[typing.Dict[str, typing.List[str]], None] = None, 335 classes_to_keep: typing.Union[typing.Dict[str, typing.List[typing.Union[str, re.Pattern]]], None] = None, 336 remove_all_attrs: typing.Union[typing.List[str], None] = None, 337 hoist_elements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None, 338 replacements: typing.Union[typing.List[typing.Tuple[str, str]], None] = None, 339 final_selectors: typing.Union[typing.List[str], None] = None, 340 ) -> str: 341 """ 342 General purpose HTML cleaning function. 343 344 filter_elements_by_descendant: { selector: (descendant_selector, text), ... } 345 Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text. 346 For example: 347 ``` 348 >>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>' 349 >>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')}) 350 <div><h2>Hello World!</h2></div> 351 ``` 352 353 delete_elements: [ selector, ... ] 354 Deletes any matching elements. 355 356 attrs_to_keep: { selector: [attribute, ...], ... } 357 Keeps only the listed attributes of elements matching the selector. 358 359 classes_to_keep: { selector: [class/pattern, ...], ... } 360 Keeps only listed classes, or classes matching the pattern, of elements matching the selector. 361 362 remove_all_attrs: [ selector, ... ] 363 Removes all attributes of matching elements. 364 365 hoist_elements: [ (parent_selector, child_selector), ... ] 366 For each pair of selectors, the parent element is replaced by the first matching child within the parent. 367 No replacement occurs if both matches are not found. 368 369 replacements: [ (pattern, replacement), ... ] 370 Performs a replacement for all regex matches. 371 372 final_selectors: [ selector, ... ] 373 Replaces html with all matched elements. 374 Useful for targeting only the necessary parts of the response. 375 Defaults to ['body'] if None. 376 """ 377 378 if (filter_elements_by_descendant is None): 379 filter_elements_by_descendant = {} 380 381 if (delete_elements is None): 382 delete_elements = [] 383 384 if (attrs_to_keep is None): 385 attrs_to_keep = {} 386 387 if (classes_to_keep is None): 388 classes_to_keep = {} 389 390 if (remove_all_attrs is None): 391 remove_all_attrs = [] 392 393 if (hoist_elements is None): 394 hoist_elements = [] 395 396 if (replacements is None): 397 replacements = [(r'\n', '')] 398 399 if (final_selectors is None): 400 final_selectors = ['body'] 401 402 document = bs4.BeautifulSoup(html, 'html.parser') 403 404 # Filter Elements by Descendant 405 for (selector, (descendant_selector, text)) in filter_elements_by_descendant.items(): 406 for element in document.select(selector): 407 descendant = element.select_one(descendant_selector) 408 if ((descendant is None) or (descendant.get_text() != text)): 409 element.decompose() 410 411 # Delete Elements 412 for selector in delete_elements: 413 for element in document.select(selector): 414 element.decompose() 415 416 # Keep Attributes 417 for (element_selector, attrs) in attrs_to_keep.items(): 418 for element in document.select(element_selector): 419 # Remove extra attributes by keeping only select attributes and replacing the existing attribute dict. 420 element.attrs = {attr: element.attrs[attr] for attr in attrs if (attr in element.attrs)} 421 422 # Keep Classes 423 for (element_selector, keep_classes) in classes_to_keep.items(): 424 for element in document.select(element_selector): 425 classes = element.get('class') 426 if ((classes is None) or (len(classes) == 0)): 427 continue 428 429 # Keep only classes listed for this selector. 430 kept = [] 431 for keep_class in keep_classes: 432 for class_name in classes: 433 if (((isinstance(keep_class, re.Pattern)) and (keep_class.fullmatch(class_name))) or (keep_class == class_name)): 434 kept.append(class_name) 435 436 element['class'] = kept # type: ignore[assignment] 437 438 # Remove All Attributes 439 for selector in remove_all_attrs: 440 for element in document.select(selector): 441 element.attrs.clear() 442 443 # Element Hoisting 444 for (parent_selector, child_selector) in hoist_elements: 445 for parent in document.select(parent_selector): 446 child = parent.select_one(child_selector) 447 if (child is None): 448 continue 449 450 parent.replace_with(child.extract()) 451 452 # Final Selectors 453 elements = [] 454 for final_selector in final_selectors: 455 for element in document.select(final_selector): 456 elements.append(element) 457 458 document_string = "".join([str(element) for element in elements]) 459 460 # Replacements 461 for (pattern, replacement) in replacements: 462 document_string = re.sub(pattern, replacement, document_string) 463 464 return document_string
General purpose HTML cleaning function.
filter_elements_by_descendant: { selector: (descendant_selector, text), ... } Removes all elements matching a selector if the selected element's specific descendant does not exist or does not have the specified text. For example:
>>> body = '<div><h2>Hello World!</h2></div><div><h2>Hello Universe!</h2></div><div><p>No h2 Element.</p></div>'
>>> clean_html(body, filter_elements_by_descendant = {'div': ('h2', 'Hello World!')})
<div><h2>Hello World!</h2></div>
delete_elements: [ selector, ... ] Deletes any matching elements.
attrs_to_keep: { selector: [attribute, ...], ... } Keeps only the listed attributes of elements matching the selector.
classes_to_keep: { selector: [class/pattern, ...], ... } Keeps only listed classes, or classes matching the pattern, of elements matching the selector.
remove_all_attrs: [ selector, ... ] Removes all attributes of matching elements.
hoist_elements: [ (parent_selector, child_selector), ... ] For each pair of selectors, the parent element is replaced by the first matching child within the parent. No replacement occurs if both matches are not found.
replacements: [ (pattern, replacement), ... ] Performs a replacement for all regex matches.
final_selectors: [ selector, ... ] Replaces html with all matched elements. Useful for targeting only the necessary parts of the response. Defaults to ['body'] if None.
466def remove_headers(response: requests.Response, headers_to_remove: typing.Set[str]) -> None: 467 """ 468 Remove headers from response and response's request. 469 """ 470 471 for headers in [response.headers, response.request.headers]: 472 for key in list(headers.keys()): # type: ignore[attr-defined] 473 if (key.strip().lower() in headers_to_remove): 474 headers.pop(key, None) # type: ignore[attr-defined]
Remove headers from response and response's request.
476def finalize_moodle_exchange(exchange: edq.net.exchange.HTTPExchange) -> edq.net.exchange.HTTPExchange: 477 """ Finalize Moodle exchanges. """ 478 479 for param in MOODLE_FINALIZE_REMOVE_PARAMS: 480 exchange.parameters.pop(param, None) 481 482 # Standardize session key. 483 if ('sesskey' in exchange.parameters): 484 exchange.parameters['sesskey'] = STANDARDIZED_SESSION_KEY 485 486 return exchange
Finalize Moodle exchanges.