quizcomp.parser.render
1import re 2import typing 3 4import edq.util.json 5import markdown_it 6import markdown_it.token 7import mdit_py_plugins.container 8import mdit_py_plugins.dollarmath 9 10import quizcomp.model.constants 11import quizcomp.parser.renderer.canvas 12import quizcomp.parser.renderer.html 13import quizcomp.parser.renderer.markdown 14import quizcomp.parser.renderer.tex 15import quizcomp.parser.renderer.text 16import quizcomp.util.html 17 18EXTRA_OPTIONS: typing.List[str] = [ 19 'table', 20] 21 22PLUGINS: typing.List[typing.Tuple[typing.Callable, typing.Dict[str, typing.Any]]] = [ 23 (mdit_py_plugins.dollarmath.dollarmath_plugin, {}), 24 (mdit_py_plugins.container.container_plugin, {'name': 'block'}), 25] 26 27HTML_TOKENS: typing.Set[str] = { 28 'html_block', 29 'html_inline', 30} 31 32def canvas( 33 tokens: typing.List[markdown_it.token.Token], 34 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 35 pretty: bool = False, 36 **kwargs: typing.Any) -> str: 37 """ Render tokens to Canvas-specific HTML. """ 38 39 if (env is None): 40 env = {} 41 42 _, options = _get_parser() 43 44 renderer = quizcomp.parser.renderer.canvas.get_renderer(options) 45 raw_html = renderer.render(tokens, options, env) 46 47 return quizcomp.util.html.clean(raw_html, pretty = pretty) 48 49def html( 50 tokens: typing.List[markdown_it.token.Token], 51 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 52 pretty: bool = False, 53 **kwargs: typing.Any) -> str: 54 """ Render tokens to HTML. """ 55 56 if (env is None): 57 env = {} 58 59 _, options = _get_parser() 60 61 renderer = quizcomp.parser.renderer.html.get_renderer(options) 62 raw_html = renderer.render(tokens, options, env) 63 64 return quizcomp.util.html.clean(raw_html, pretty = pretty) 65 66def md( 67 tokens: typing.List[markdown_it.token.Token], 68 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 69 **kwargs: typing.Any) -> str: 70 """ Render tokens to Markdown. """ 71 72 if (env is None): 73 env = {} 74 75 _, options = _get_parser() 76 77 renderer = quizcomp.parser.renderer.markdown.get_renderer(options) 78 content = renderer.render(tokens, options, env) 79 80 return content.strip() 81 82def tex( 83 tokens: typing.List[markdown_it.token.Token], 84 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 85 **kwargs: typing.Any) -> str: 86 """ Render tokens to TeX. """ 87 88 if (env is None): 89 env = {} 90 91 _, options = _get_parser() 92 93 renderer = quizcomp.parser.renderer.tex.get_renderer(options) 94 content = renderer.render(tokens, options, env) 95 96 return content.strip() 97 98def text( 99 tokens: typing.List[markdown_it.token.Token], 100 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 101 **kwargs: typing.Any) -> str: 102 """ Render tokens to text. """ 103 104 if (env is None): 105 env = {} 106 107 _, options = _get_parser() 108 109 renderer = quizcomp.parser.renderer.text.get_renderer(options) 110 content = renderer.render(tokens, options, env) 111 112 return content.strip() 113 114def render( 115 format: quizcomp.model.constants.Format, 116 tokens: typing.List[markdown_it.token.Token], 117 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 118 **kwargs: typing.Any) -> str: 119 """ Render tokens to the specified format. """ 120 121 if (env is None): 122 env = {} 123 124 render_function = globals().get(format.value, None) 125 if (render_function is None): 126 raise ValueError(f"Could not find render function: 'quizcomp.parser.render.{format.value}'.") 127 128 return str(render_function(tokens, env = env, **kwargs)) 129 130def _get_parser() -> typing.Tuple[markdown_it.MarkdownIt, markdown_it.utils.OptionsDict]: 131 """ Get the standard parser and options. """ 132 133 parser = markdown_it.MarkdownIt('commonmark') 134 135 for option in EXTRA_OPTIONS: 136 parser.enable(option) 137 138 for (plugin, options) in PLUGINS: 139 parser.use(plugin, **options) 140 141 return parser, parser.options 142 143def _clean_text(content: str) -> str: 144 """ Do some basic cleaning on text before parsing. """ 145 146 # Remove carriage returns. 147 content = content.replace("\r", '') 148 149 # Trim whitespace. 150 content = content.strip() 151 152 return content 153 154# Returns (transformed text, tokens). 155def _parse_text(raw_text: str) -> typing.Tuple[str, typing.List[markdown_it.token.Token]]: 156 """ Parse the text and returned the cleaned text and resulting parsed document. """ 157 158 clean_text = _clean_text(raw_text) 159 160 if (len(clean_text) == 0): 161 return '', [] 162 163 parser, _ = _get_parser() 164 165 tokens = parser.parse(clean_text) 166 tokens = _post_process(tokens) 167 168 return clean_text.strip(), tokens 169 170def _post_process(tokens: typing.List[markdown_it.token.Token]) -> typing.List[markdown_it.token.Token]: 171 """ 172 Post-process the token stream. 173 This allows us to edit the AST without needing the change the parser. 174 """ 175 176 tokens = _add_root_block(tokens) 177 tokens = _process_placeholders(tokens) 178 tokens = _process_style(tokens) 179 tokens = _process_html(tokens) 180 tokens = _remove_empty_tokens(tokens) 181 182 return tokens 183 184def _add_root_block(tokens: typing.List[markdown_it.token.Token]) -> typing.List[markdown_it.token.Token]: 185 """ 186 Add a root block element to the document. 187 """ 188 189 if (len(tokens) == 0): 190 return [] 191 192 open_token = markdown_it.token.Token('container_block_open', 'div', 1) 193 open_token.block = True 194 open_token.attrJoin('class', 'qg-root-block') 195 open_token.meta[quizcomp.parser.common.TOKEN_META_KEY_ROOT] = True 196 197 if (tokens[0].map is None): 198 open_token.map = [] 199 else: 200 open_token.map = list(tokens[0].map) 201 202 close_token = markdown_it.token.Token('container_block_close', 'div', -1) 203 204 return [open_token] + tokens + [close_token] 205 206def _process_style( 207 tokens: typing.Union[typing.List[markdown_it.token.Token], None], 208 containing_block: typing.Union[markdown_it.token.Token, None] = None, 209 ) -> typing.List[markdown_it.token.Token]: 210 """ 211 Locate any style nodes, parse them, remove them, and hoist their content to the containing block. 212 """ 213 214 if (tokens is None): 215 return [] 216 217 remove_indexes = [] 218 for (i, token) in enumerate(tokens): 219 # If this is a block, then mark it as the current block. 220 # Any discovered style get's hoisted to the containing block. 221 if (token.type == 'container_block_open'): 222 containing_block = token 223 elif ((token.type in HTML_TOKENS) and (token.content.strip().startswith('<style>'))): 224 # Style nodes are HTML with a 'style' tag. 225 226 if (containing_block is None): 227 raise ValueError("Found a style node that does not have a containing block.") 228 229 style = _process_style_content(token.content) 230 containing_block.meta[quizcomp.parser.common.TOKEN_META_KEY_STYLE] = style 231 232 # Mark this token for removal. 233 remove_indexes.append(i) 234 235 # Check all children. 236 if (_has_children(token)): 237 token.children = _process_style(token.children, containing_block = containing_block) 238 239 for remove_index in sorted(list(set(remove_indexes)), reverse = True): 240 tokens.pop(remove_index) 241 242 return tokens 243 244def _process_style_content(raw_content: str) -> typing.Dict[str, typing.Any]: 245 """ Get the style content from some text. """ 246 247 raw_content = raw_content.strip() 248 249 # Get content without tags ('<style>', '</style>'). 250 content = re.sub(r'\s+', ' ', raw_content) 251 content = re.sub(r'^<style>(.*)</style>$', r'\1', content).strip() 252 253 # Ignore empty style. 254 if (len(content) == 0): 255 return {} 256 257 # If the content does not start with a '{', then assume the braces were left out and add them. 258 # We will also ignore content that starts with a '[' (a JSON list), that will be handled later. 259 if (content[0] not in ['{', '[']): 260 content = "{%s}" % (content) 261 262 try: 263 style = edq.util.json.loads(content) 264 if (not isinstance(style, dict)): 265 raise ValueError(f"Style is not a JSON object, found: '{type(style)}'.") 266 except Exception as ex: 267 raise ValueError(('Failed to load style tag.' # pylint: disable=raise-missing-from 268 + ' Style content must be a JSON object (start/end braces may be omitted).' 269 + f" Original exception message: '{ex}'." 270 + f" Found:\n---\n{raw_content}\n---")) 271 272 return style 273 274def _remove_empty_tokens(tokens: typing.Union[typing.List[markdown_it.token.Token], None]) -> typing.List[markdown_it.token.Token]: 275 """ 276 Remove any inline's without text or blocks without children. 277 """ 278 279 if (tokens is None): 280 return [] 281 282 # Keep looping until nothing is removed. 283 while True: 284 remove_indexes = [] 285 for (i, token) in enumerate(tokens): 286 # Remove empty leaf content nodes. 287 if ((token.type in quizcomp.parser.common.CONTENT_NODES) and (token.content == '')): 288 remove_indexes.append(i) 289 290 # Check children for removal. 291 if (_has_children(token)): 292 token.children = _remove_empty_tokens(token.children) 293 294 # Remove nodes that have been emptied out. 295 if (len(token.children) == 0): 296 remove_indexes.append(i) 297 298 # Check for empty containers. 299 # Look for this token being the open, and the next token being the close. 300 if ((i < (len(tokens) - 1)) and (token.type.endswith('_open')) and (tokens[i + 1].type.endswith('_close'))): 301 next_token = tokens[i + 1] 302 base_type = re.sub('_open$', '', token.type) 303 next_base_type = re.sub('_close$', '', next_token.type) 304 305 # Remove if types match and neither has any kids (close should never have kids). 306 if ((base_type == next_base_type) and (not _has_children(token)) and (not _has_children(next_token))): 307 remove_indexes += [i, i + 1] 308 309 for remove_index in sorted(list(set(remove_indexes)), reverse = True): 310 tokens.pop(remove_index) 311 312 if (len(remove_indexes) == 0): 313 break 314 315 return tokens 316 317def _process_placeholders(tokens: typing.Union[typing.List[markdown_it.token.Token], None]) -> typing.List[markdown_it.token.Token]: 318 """ 319 Find any placeholder HTML tags and replace them with a placeholder token. 320 plceholder tags must either be an HTML block or inline with the same parent. 321 """ 322 323 if (tokens is None): 324 return [] 325 326 remove_indexes = [] 327 328 # Use a non-standard loop so that we can manually advance the index within the loop. 329 i = -1 330 while (i < (len(tokens) - 1)): 331 i += 1 332 token = tokens[i] 333 334 if (token.type == 'html_block'): 335 if ((not token.content) or (not token.content.strip().startswith('<placeholder'))): 336 continue 337 338 # Replace the HTML block token with a placeholder token. 339 tokens[i] = _create_placeholder_token(token) 340 elif (token.type == 'html_inline'): 341 if ((not token.content) or (not token.content.strip().startswith('<placeholder'))): 342 continue 343 344 open_tag_index = i 345 close_tag_index = None 346 347 # Look for the close tag at this same level (under the same parent). 348 for j in range(i + 1, len(tokens)): 349 other_token = tokens[j] 350 if (other_token.type != 'html_inline'): 351 continue 352 353 if ((not other_token.content) or (not other_token.content.strip() == '</placeholder>')): 354 continue 355 356 close_tag_index = j 357 break 358 359 if (close_tag_index is None): 360 raise ValueError("Could not find closing tag for <placeholder>.") 361 362 if ((close_tag_index - open_tag_index) < 2): 363 raise ValueError("Did not find any content inside a <placeholder> tag.") 364 365 if ((close_tag_index - open_tag_index) > 2): 366 raise ValueError("Found too much content inside a <placeholder> tag, it shoud have only plain text.") 367 368 text_token_index = open_tag_index + 1 369 text_token = tokens[text_token_index] 370 if (text_token.type != 'text'): 371 raise ValueError("Found non-text content inside a <placeholder> tag, it shoud have only plain text.") 372 373 # All tokens (open, content/label, close) are accounted for. 374 # Replace the content node and remove the open/close tags. 375 tokens[text_token_index] = _create_placeholder_token(text_token) 376 remove_indexes += [open_tag_index, close_tag_index] 377 378 # Advance to the close token. 379 i = close_tag_index 380 381 if (_has_children(token)): 382 token.children = _process_placeholders(token.children) 383 384 for remove_index in sorted(list(set(remove_indexes)), reverse = True): 385 tokens.pop(remove_index) 386 387 return tokens 388 389def _create_placeholder_token(token: markdown_it.token.Token) -> markdown_it.token.Token: 390 """ Create a tag for answer placeholders. """ 391 392 # Fetch the label in the tag. 393 label = re.sub(r'\s+', ' ', token.content.strip()) 394 label = re.sub(r'^<placeholder.*>(.*)</placeholder>$', r'\1', label).strip() 395 if (len(label) == 0): 396 raise ValueError("Found an empty '<placeholder>' tag.") 397 398 return markdown_it.token.Token( 399 type = 'placeholder', tag = '', nesting = 0, 400 map = token.map, content = label) 401 402def _process_html(tokens: typing.Union[typing.List[markdown_it.token.Token], None]) -> typing.List[markdown_it.token.Token]: 403 """ 404 Remove or replacec all HTML tags. 405 This will recursively descend to find all tags. 406 Line breaks will be replaced with hard breaks. 407 """ 408 409 if (tokens is None): 410 return [] 411 412 remove_indexes = [] 413 for (i, token) in enumerate(tokens): 414 if (token.type in HTML_TOKENS): 415 if (token.content.strip().startswith('<br')): 416 tokens[i] = markdown_it.token.Token( 417 type = 'hardbreak', tag = 'br', nesting = 0, 418 map = token.map) 419 else: 420 remove_indexes.append(i) 421 422 if (_has_children(token)): 423 token.children = _process_html(token.children) 424 425 for remove_index in sorted(list(set(remove_indexes)), reverse = True): 426 tokens.pop(remove_index) 427 428 return tokens 429 430def _has_children(token: markdown_it.token.Token) -> bool: 431 """ Check if the given token has children. """ 432 433 return ((token.children is not None) and (len(token.children) > 0))
EXTRA_OPTIONS: List[str] =
['table']
PLUGINS: List[Tuple[Callable, Dict[str, Any]]] =
[(<function dollarmath_plugin>, {}), (<function container_plugin>, {'name': 'block'})]
HTML_TOKENS: Set[str] =
{'html_inline', 'html_block'}
def
canvas( tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, pretty: bool = False, **kwargs: Any) -> str:
33def canvas( 34 tokens: typing.List[markdown_it.token.Token], 35 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 36 pretty: bool = False, 37 **kwargs: typing.Any) -> str: 38 """ Render tokens to Canvas-specific HTML. """ 39 40 if (env is None): 41 env = {} 42 43 _, options = _get_parser() 44 45 renderer = quizcomp.parser.renderer.canvas.get_renderer(options) 46 raw_html = renderer.render(tokens, options, env) 47 48 return quizcomp.util.html.clean(raw_html, pretty = pretty)
Render tokens to Canvas-specific HTML.
def
html( tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, pretty: bool = False, **kwargs: Any) -> str:
50def html( 51 tokens: typing.List[markdown_it.token.Token], 52 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 53 pretty: bool = False, 54 **kwargs: typing.Any) -> str: 55 """ Render tokens to HTML. """ 56 57 if (env is None): 58 env = {} 59 60 _, options = _get_parser() 61 62 renderer = quizcomp.parser.renderer.html.get_renderer(options) 63 raw_html = renderer.render(tokens, options, env) 64 65 return quizcomp.util.html.clean(raw_html, pretty = pretty)
Render tokens to HTML.
def
md( tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, **kwargs: Any) -> str:
67def md( 68 tokens: typing.List[markdown_it.token.Token], 69 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 70 **kwargs: typing.Any) -> str: 71 """ Render tokens to Markdown. """ 72 73 if (env is None): 74 env = {} 75 76 _, options = _get_parser() 77 78 renderer = quizcomp.parser.renderer.markdown.get_renderer(options) 79 content = renderer.render(tokens, options, env) 80 81 return content.strip()
Render tokens to Markdown.
def
tex( tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, **kwargs: Any) -> str:
83def tex( 84 tokens: typing.List[markdown_it.token.Token], 85 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 86 **kwargs: typing.Any) -> str: 87 """ Render tokens to TeX. """ 88 89 if (env is None): 90 env = {} 91 92 _, options = _get_parser() 93 94 renderer = quizcomp.parser.renderer.tex.get_renderer(options) 95 content = renderer.render(tokens, options, env) 96 97 return content.strip()
Render tokens to TeX.
def
text( tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, **kwargs: Any) -> str:
99def text( 100 tokens: typing.List[markdown_it.token.Token], 101 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 102 **kwargs: typing.Any) -> str: 103 """ Render tokens to text. """ 104 105 if (env is None): 106 env = {} 107 108 _, options = _get_parser() 109 110 renderer = quizcomp.parser.renderer.text.get_renderer(options) 111 content = renderer.render(tokens, options, env) 112 113 return content.strip()
Render tokens to text.
def
render( format: quizcomp.model.constants.Format, tokens: List[markdown_it.token.Token], env: Optional[Dict[str, Any]] = None, **kwargs: Any) -> str:
115def render( 116 format: quizcomp.model.constants.Format, 117 tokens: typing.List[markdown_it.token.Token], 118 env: typing.Union[typing.Dict[str, typing.Any], None] = None, 119 **kwargs: typing.Any) -> str: 120 """ Render tokens to the specified format. """ 121 122 if (env is None): 123 env = {} 124 125 render_function = globals().get(format.value, None) 126 if (render_function is None): 127 raise ValueError(f"Could not find render function: 'quizcomp.parser.render.{format.value}'.") 128 129 return str(render_function(tokens, env = env, **kwargs))
Render tokens to the specified format.