class Canon::Diff::DiffClassifier
-
Normative: Semantic content differences (affect equivalence)
2. Content formatting: Whitespace differences in content (non-normative when normalized)
1. Serialization formatting: XML syntax differences (always non-normative)
Classification hierarchy (three distinct kinds of differences):
based on the match options in effect
Classifies DiffNodes as normative (affects equivalence) or informative (doesn’t affect equivalence)
def classify(diff_node)
-
(DiffNode)- The same diff node with normative/formatting attributes set
Parameters:
-
diff_node(DiffNode) -- The diff node to classify
def classify(diff_node) # FIRST: Check for XML serialization-level formatting differences # These are ALWAYS non-normative (formatting-only) regardless of match options # Examples: self-closing tags (<tag/>) vs explicit closing tags (<tag></tag>) # # EXCEPTION: If the text node is inside a whitespace-sensitive element # (:preserve or :collapse), don't dismiss as serialization formatting # because whitespace presence is meaningful in those elements. if !inside_whitespace_sensitive_element?(diff_node) && XmlSerializationFormatter.serialization_formatting?(diff_node) diff_node.formatting = true diff_node.normative = false return diff_node end # SECOND: Handle content-level formatting for text_content with :normalize behavior # When text_content is :normalize and the difference is formatting-only, # it should be marked as non-normative (informative) # This ensures that verbose and non-verbose modes give consistent results # # EXCEPTION: If the text node is inside a PRESERVE whitespace element # (like <pre>, <code>, <textarea> in HTML), don't apply formatting detection # because whitespace should be preserved exactly in these elements. # Note: COLLAPSE elements like <p> DO get formatting detection because # their whitespace IS normalized (differences are formatting-only). # # This check must come BEFORE normative_dimension? is called, # because normative_dimension? returns true for text_content: :normalize # (since the dimension affects equivalence), which would prevent formatting # detection from being applied. if diff_node.dimension == :text_content && profile.behavior_for(:text_content) == :normalize && !inside_preserve_element?(diff_node) && formatting_only_diff?(diff_node) diff_node.formatting = true diff_node.normative = false return diff_node end # :whitespace_adjacency is a report-only re-label of an # asymmetric whitespace mismatch emitted by ChildRealignment's # two-cursor walk. Equivalence behaviour is unchanged — the # underlying mismatch is normative regardless of match options. if diff_node.dimension == :whitespace_adjacency diff_node.formatting = false diff_node.normative = true return diff_node end # :comments diffs from asymmetric comment nodes intentionally # fall through to profile.normative_dimension? below. Unlike # :whitespace_adjacency (always normative), the classification # of comment diffs respects the :comments match option: # :strict → normative, :ignore → informative. This is by # design — callers can control whether asymmetric comments # affect equivalence via the match profile. # THIRD: Determine if this dimension is normative based on CompareProfile # This respects the policy settings (strict/normalize/ignore) is_normative = profile.normative_dimension?(diff_node.dimension) # FOURTH: Check if FormattingDetector should be consulted for non-normative dimensions # Only check for formatting-only when dimension is NOT normative # This ensures strict mode differences remain normative should_check_formatting = !is_normative && profile.supports_formatting_detection?(diff_node.dimension) # If we should check formatting, see if it's formatting-only if should_check_formatting && formatting_only_diff?(diff_node) diff_node.formatting = true diff_node.normative = false return diff_node end # FIFTH: Apply the normative determination from CompareProfile diff_node.formatting = false diff_node.normative = is_normative diff_node end
def classify_all(diff_nodes)
-
(Array- The same diff nodes with normative attributes set)
Parameters:
-
diff_nodes(Array) -- The diff nodes to classify
def classify_all(diff_nodes) diff_nodes.each { |node| classify(node) } end
def extract_text_content(node)
def extract_text_content(node) return nil if node.nil? Canon::Comparison::NodeInspector.text_content(node) rescue StandardError nil end
def formatting_only_diff?(diff_node)
-
(Boolean)- true if formatting-only
Parameters:
-
diff_node(DiffNode) -- The diff node to check
def formatting_only_diff?(diff_node) # Only apply formatting detection to actual text content differences # If the nodes are not text nodes (e.g., element nodes), don't apply formatting detection node1 = diff_node.node1 node2 = diff_node.node2 # Check if both nodes are text nodes # If not, this is not a formatting-only difference return false unless text_node?(node1) && text_node?(node2) text1 = extract_text_content(diff_node.node1) text2 = extract_text_content(diff_node.node2) # For text_content dimension, use normalized text comparison # This handles cases like "" vs " " (both normalize to "") if diff_node.dimension == :text_content normalized_equivalent?(text1, text2) else FormattingDetector.formatting_only?(text1, text2) end end
def initialize(match_options)
-
match_options(Canon::Comparison::ResolvedMatchOptions) -- The match options
def initialize(match_options) @match_options = match_options # Use the compare_profile from ResolvedMatchOptions if available (e.g., HtmlCompareProfile) # Otherwise create a base CompareProfile @profile = if match_options.is_a?(Canon::Comparison::ResolvedMatchOptions) && match_options.compare_profile match_options.compare_profile else Canon::Comparison::CompareProfile.new(match_options) end end
def inside_preserve_element?(diff_node)
-
(Boolean)- true if inside a preserve whitespace element
Parameters:
-
diff_node(DiffNode) -- The diff node to check
def inside_preserve_element?(diff_node) node = diff_node.node1 || diff_node.node2 return false unless node match_opts = @match_options.options parent = node.parent return false unless parent classification = Canon::Comparison::WhitespaceSensitivity.classify_element( parent, match_opts ) classification == :preserve end
def inside_whitespace_sensitive_element?(diff_node)
-
(Boolean)- true if whitespace is preserved for this element
Parameters:
-
diff_node(DiffNode) -- The diff node to check
def inside_whitespace_sensitive_element?(diff_node) node = diff_node.node1 || diff_node.node2 return false unless node # HTML: whitespace between inline element siblings is significant if Canon::Comparison::WhitespaceSensitivity.inline_whitespace_significant?(node) return true end # HTML: non-breaking space (U+00A0) is never insignificant text = Canon::Comparison::NodeInspector.text_content(node) if text && Canon::Comparison::WhitespaceSensitivity.contains_nbsp?(text) return true end return false unless Canon::XmlParsing.element?(node) || node.is_a?(Canon::Xml::Node) parent = node.parent return false unless parent match_opts = @match_options.options Canon::Comparison::WhitespaceSensitivity.whitespace_preserved?(parent, match_opts) end
def normalized_equivalent?(text1, text2)
-
(Boolean)- true if normalized texts are equivalent
Parameters:
-
text2(String, nil) -- Second text -
text1(String, nil) -- First text
def normalized_equivalent?(text1, text2) return false if text1.nil? && text2.nil? return false if text1.nil? || text2.nil? # Use MatchOptions.normalize_text for consistency normalized1 = Canon::Comparison::MatchOptions.normalize_text(text1) normalized2 = Canon::Comparison::MatchOptions.normalize_text(text2) # If normalized texts are equivalent but originals are different, # it's a formatting-only difference normalized1 == normalized2 && text1 != text2 end
def text_node?(node)
def text_node?(node) return false if node.nil? return true if node.is_a?(String) Canon::Comparison::NodeInspector.text_node?(node) end