Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions lib/nokolexbor/node.rb
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,32 @@ class Node

LOOKS_LIKE_XPATH = %r{^(\./|/|\.\.|\.$)}

CDATA_GUARD_PATTERNS = [
%r{\A(?<before>[[:space:]]*)<!\[CDATA\[(?<text>.*)\]\]>(?<after>[[:space:]]*)\z}m,
%r{\A(?<before>[[:space:]]*)//[[:blank:]]*<!\[CDATA\[(?<text>.*)//[[:blank:]]*\]\]>(?<after>[[:space:]]*)\z}m,
%r{\A(?<before>[[:space:]]*)/\*[[:blank:]]*<!\[CDATA\[[[:blank:]]*\*/(?<text>.*)/\*[[:blank:]]*\]\]>[[:blank:]]*\*/(?<after>[[:space:]]*)\z}m,
%r{\A(?<before>[[:space:]]*)/\*[[:blank:]]*<!\[CDATA\[[[:blank:]]*/\*[[:blank:]]*\*/(?<text>.*)/\*[[:blank:]]*\]\]>[[:blank:]]*/\*[[:blank:]]*\*/(?<after>[[:space:]]*)\z}m,
].freeze
private_constant :CDATA_GUARD_PATTERNS

# Return this node's text with one complete legacy CDATA guard removed.
#
# This does not modify the node or change {#content}. Incomplete, mismatched,
# and embedded guards are returned unchanged. Whitespace is preserved; use
# +node.unwrap_cdata_text.strip+ to remove surrounding whitespace as well.
#
# @return [String]
def unwrap_cdata_text
raw_text = content

CDATA_GUARD_PATTERNS.each do |pattern|
match = pattern.match(raw_text)
return match[:before] + match[:text] + match[:after] if match
end

raw_text
end

# @return true if this is a {Comment}
def comment?
type == COMMENT_NODE
Expand Down
61 changes: 61 additions & 0 deletions spec/node_spec.rb
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,67 @@
_{ node.content = 1 }.must_raise TypeError
end

describe 'unwrap_cdata_text' do
def script_with(content)
Nokolexbor::HTML("<script>#{content}</script>").at_css('script')
end

it 'removes recognized complete CDATA guards and preserves whitespace' do
wrapped_text = {
"<![CDATA[\n payload\n]]>" => "\n payload\n",
"//<![CDATA[\n payload\n//]]>" => "\n payload\n",
"/* <![CDATA[ */\n payload\n/* ]]> */" => "\n payload\n",
"/*<![CDATA[*/\n payload\n/*]]>*/" => "\n payload\n",
"/*<![CDATA[/* */\n payload\n/*]]>/* */" => "\n payload\n",
}

wrapped_text.each do |text, expected|
_(script_with(text).unwrap_cdata_text).must_equal expected
end
end

it 'preserves whitespace from the node content' do
node = script_with('')
node.content = " \t<![CDATA[payload]]>\r\n"

_(node.unwrap_cdata_text).must_equal " \tpayload\r\n"
end

it 'composes with String#strip for generic surrounding whitespace cleanup' do
text_values = [
" payload ",
"<![CDATA[\n payload\n]]>",
" /* <![CDATA[ */\n payload\n/* ]]> */ ",
]

text_values.each do |text|
_(script_with(text).unwrap_cdata_text.strip).must_equal 'payload'
end
end

it 'does not change the node content' do
node = script_with('<![CDATA[payload]]>')

_(node.unwrap_cdata_text).must_equal 'payload'
_(node.content).must_equal '<![CDATA[payload]]>'
end

it 'preserves text without a complete matching outer wrapper' do
text_values = [
" payload ",
'<![CDATA[payload',
'payload]]>',
'//<![CDATA[payload]]>',
'before <![CDATA[payload]]> after',
'/* <![CDATA[ */payload/*]]>/* */',
]

text_values.each do |text|
_(script_with(text).unwrap_cdata_text).must_equal text
end
end
end

describe 'attr' do
before do
@doc = Nokolexbor::HTML('<div class="a" checked></div>')
Expand Down
Loading