Improve first par detection without hpricot

[user/henk/code/ruby/rbot.git] / lib / rbot / core / utils / utils.rb
diff --git a/lib/rbot/core/utils/utils.rb b/lib/rbot/core/utils/utils.rb

index 4105fa12db611c0988d5f91d0757c458d44e2168..e3392f1fafadad95c38d113d1b1bc540c6585f7e 100644 (file)
--- a/lib/rbot/core/utils/utils.rb
+++ b/lib/rbot/core/utils/utils.rb
@@ -23,6 +23,7 @@ rescue LoadError
      'raquo' => '»',
      'quot' => '"',
      'apos' => '\'',
+    'deg' => '°',
      'micro' => 'µ',
      'copy' => '©',
      'trade' => '™',
@@ -32,6 +33,7 @@ rescue LoadError
      'gt' => '>',
      'hellip' => '…',
      'nbsp' => ' ',
+    'ndash' => '–',
      'Agrave' => 'À',
      'Aacute' => 'Á',
      'Acirc' => 'Â',
@@ -125,7 +127,7 @@ rescue LoadError
  
          # Some blogging and forum platforms use spans or divs with a 'body' or 'message' or 'text' in their class
          # to mark actual text
-        AFTER_PAR1_REGEX = /<\w+\s+[^>]*(?:body|message|text)[^>]*>.*?<\/?(?:p|div|html|body|table|td|tr)(?:\s+[^>]*)?>/im
+        AFTER_PAR1_REGEX = /<\w+\s+[^>]*(?:body|message|text|post)[^>]*>.*?<\/?(?:p|div|html|body|table|td|tr)(?:\s+[^>]*)?>/im
  
          # At worst, we can try stuff which is comprised between two <br>
          AFTER_PAR2_REGEX = /<br(?:\s+[^>]*)?\/?>.*?<\/?(?:br|p|div|html|body|table|td|tr)(?:\s+[^>]*)?\/?>/im
@@ -336,11 +338,21 @@ module ::Irc
      # Decode HTML entities in the String _str_, using HTMLEntities if the
      # package was found, or UNESCAPE_TABLE otherwise.
      #
-    def Utils.decode_html_entities(str)
-      if defined? ::HTMLEntities
-        return HTMLEntities.decode_entities(str)
+
+    if defined? ::HTMLEntities
+      if ::HTMLEntities.respond_to? :decode_entities
+        def Utils.decode_html_entities(str)
+          return HTMLEntities.decode_entities(str)
+        end
        else
-        str.gsub(/(&(.+?);)/) {
+        @@html_entities = HTMLEntities.new
+        def Utils.decode_html_entities(str)
+          return @@html_entities.decode str
+        end
+      end
+    else
+      def Utils.decode_html_entities(str)
+        return str.gsub(/(&(.+?);)/) {
            symbol = $2
            # remove the 0-paddng from unicode integers
            if symbol =~ /^#(\d+)$/
@@ -481,7 +493,11 @@ module ::Irc
  
      # HTML first par grabber without hpricot
      def Utils.ircify_first_html_par_woh(xml_org, opts={})
-      xml = xml_org.gsub(/<!--.*?-->/m, '').gsub(/<script(?:\s+[^>]*)?>.*?<\/script>/im, "").gsub(/<style(?:\s+[^>]*)?>.*?<\/style>/im, "")
+      xml = xml_org.gsub(/<!--.*?-->/m,
+                         "").gsub(/<script(?:\s+[^>]*)?>.*?<\/script>/im,
+                         "").gsub(/<style(?:\s+[^>]*)?>.*?<\/style>/im,
+                         "").gsub(/<select(?:\s+[^>]*)?>.*?<\/select>/im,
+                         "")
  
        strip = opts[:strip]
        strip = Regexp.new(/^#{Regexp.escape(strip)}/) if strip.kind_of?(String)