* Extracting container for allowed tags & attrs Prior to this commit, we had several different locations in which we specified ALLOWED_TAGS and ALLOWED_ATTRIBUTES for HTML rendering and sanitization. Curious to see how these either intersected or didn't, I opted to create a container module that allows for us to more readily normalize these allowed tags and attributes. It's possible that we won't do any normalization, but this work helps make that easier. Ideally, I'd love us to contextualize "why did we choose the tags/attributes we chose?" But for now, I think consolidating these tags and attributes will help make adding a `details` and `summary` tag easier. This relates to forem/rfcs#296 See [Google Sheet][1] for analysis of what tags/attributes are used, the intersection and union. [1]:https://docs.google.com/spreadsheets/d/1yj-a1qus1o0o4cj-_gOMP5yteeg-_f3s5z7kvK0Y7RM/edit#gid=0 * Fixing misnamed constant * Fixing misnamed constant * Extracting additional HtmlRendering use cases * Adding comparative documentation for HTML tags * Fixing broken parameter signature * Moving constants into MarkdownProcessor
138 lines
4.8 KiB
Ruby
138 lines
4.8 KiB
Ruby
module MarkdownProcessor
|
|
class Parser
|
|
include ApplicationHelper
|
|
|
|
BAD_XSS_REGEX = [
|
|
/src=["'](data|&)/i,
|
|
%r{data:text/html[,;][\sa-z0-9]*}i,
|
|
].freeze
|
|
|
|
WORDS_READ_PER_MINUTE = 275.0
|
|
|
|
def initialize(content, source: nil, user: nil)
|
|
@content = content
|
|
@source = source
|
|
@user = user
|
|
end
|
|
|
|
def finalize(link_attributes: {})
|
|
options = { hard_wrap: true, filter_html: false, link_attributes: link_attributes }
|
|
renderer = Redcarpet::Render::HTMLRouge.new(options)
|
|
markdown = Redcarpet::Markdown.new(renderer, Constants::Redcarpet::CONFIG)
|
|
catch_xss_attempts(@content)
|
|
code_tag_content = convert_code_tags_to_triple_backticks(@content)
|
|
escaped_content = escape_liquid_tags_in_codeblock(code_tag_content)
|
|
html = markdown.render(escaped_content)
|
|
sanitized_content = sanitize_rendered_markdown(html)
|
|
begin
|
|
liquid_tag_options = { source: @source, user: @user }
|
|
|
|
# NOTE: [@rhymes] liquid 5.0.0 does not support ActiveSupport::SafeBuffer,
|
|
# a String substitute, hence we force the conversion before passing it to Liquid::Template.
|
|
# See <https://github.com/Shopify/liquid/issues/1390>
|
|
parsed_liquid = Liquid::Template.parse(sanitized_content.to_str, liquid_tag_options)
|
|
|
|
html = markdown.render(parsed_liquid.render)
|
|
rescue Liquid::SyntaxError => e
|
|
html = e.message
|
|
end
|
|
|
|
parse_html(html)
|
|
end
|
|
|
|
def calculate_reading_time
|
|
word_count = @content.split(/\W+/).count
|
|
(word_count / WORDS_READ_PER_MINUTE).ceil
|
|
end
|
|
|
|
def evaluate_markdown(allowed_tags: MarkdownProcessor::AllowedTags::MARKDOWN_PROCESSOR_DEFAULT)
|
|
return if @content.blank?
|
|
|
|
renderer = Redcarpet::Render::HTMLRouge.new(hard_wrap: true, filter_html: false)
|
|
markdown = Redcarpet::Markdown.new(renderer, Constants::Redcarpet::CONFIG)
|
|
ActionController::Base.helpers.sanitize(markdown.render(@content),
|
|
tags: allowed_tags,
|
|
attributes: MarkdownProcessor::AllowedAttributes::MARKDOWN_PROCESSOR)
|
|
end
|
|
|
|
def evaluate_limited_markdown(allowed_tags: MarkdownProcessor::AllowedTags::MARKDOWN_PROCESSOR_LIMITED)
|
|
evaluate_markdown(allowed_tags: allowed_tags)
|
|
end
|
|
|
|
# rubocop:disable Layout/LineLength
|
|
def evaluate_inline_limited_markdown(allowed_tags: MarkdownProcessor::AllowedTags::MARKDOWN_PROCESSOR_INLINE_LIMITED)
|
|
evaluate_markdown(allowed_tags: allowed_tags)
|
|
end
|
|
# rubocop:enable Layout/LineLength
|
|
|
|
def evaluate_listings_markdown(allowed_tags: MarkdownProcessor::AllowedTags::MARKDOWN_PROCESSOR_LISTINGS)
|
|
evaluate_markdown(allowed_tags: allowed_tags)
|
|
end
|
|
|
|
def tags_used
|
|
return [] if @content.blank?
|
|
|
|
cleaned_parsed = escape_liquid_tags_in_codeblock(@content)
|
|
tags = []
|
|
liquid_tag_options = { source: @source, user: @user }
|
|
Liquid::Template.parse(cleaned_parsed, liquid_tag_options).root.nodelist.each do |node|
|
|
tags << node.class if node.class.superclass.to_s == LiquidTagBase.to_s
|
|
end
|
|
tags.uniq
|
|
rescue Liquid::SyntaxError
|
|
[]
|
|
end
|
|
|
|
def catch_xss_attempts(markdown)
|
|
return unless markdown.match?(Regexp.union(BAD_XSS_REGEX))
|
|
|
|
raise ArgumentError, "Invalid markdown detected!"
|
|
end
|
|
|
|
def escape_liquid_tags_in_codeblock(content)
|
|
# Escape codeblocks, code spans, and inline code
|
|
content.gsub(/[[:space:]]*~{3}.*?~{3}|[[:space:]]*`{3}.*?`{3}|`{2}.+?`{2}|`{1}.+?`{1}/m) do |codeblock|
|
|
codeblock.gsub!("{% endraw %}", "{----% endraw %----}")
|
|
codeblock.gsub!("{% raw %}", "{----% raw %----}")
|
|
if codeblock.match?(/[[:space:]]*`{3}/)
|
|
"\n{% raw %}\n#{codeblock}\n{% endraw %}\n"
|
|
else
|
|
"{% raw %}#{codeblock}{% endraw %}"
|
|
end
|
|
end
|
|
end
|
|
|
|
def convert_code_tags_to_triple_backticks(content)
|
|
# return content if there is not a <code> tag
|
|
return content unless /^<code>$/.match?(content)
|
|
|
|
# return content if there is a <pre> and <code> tag
|
|
return content if /<code>/.match?(content) && /<pre>/.match?(content)
|
|
|
|
# Convert all multiline code tags to triple backticks
|
|
content.gsub(%r{^</?code>$}, "\n```\n")
|
|
end
|
|
|
|
private
|
|
|
|
def parse_html(html)
|
|
return html if html.blank?
|
|
|
|
Html::Parser
|
|
.new(html)
|
|
.remove_nested_linebreak_in_list
|
|
.prefix_all_images
|
|
.wrap_all_images_in_links
|
|
.add_control_class_to_codeblock
|
|
.add_control_panel_to_codeblock
|
|
.add_fullscreen_button_to_panel
|
|
.wrap_all_tables
|
|
.remove_empty_paragraphs
|
|
.escape_colon_emojis_in_codeblock
|
|
.unescape_raw_tag_in_codeblocks
|
|
.wrap_all_figures_with_tags
|
|
.wrap_mentions_with_links
|
|
.html
|
|
end
|
|
end
|
|
end
|