Automated merge.

[wget] / src / html-url.c
diff --git a/src/html-url.c b/src/html-url.c

index 3ab7f7fe02127410bf8f3781874cf74fc701e203..9b515432a8122550b9beaca988bce6e99d1c16dd 100644 (file)
--- a/src/html-url.c
+++ b/src/html-url.c
@@ -1,6 +1,6 @@
  /* Collect URLs from HTML source.
     Copyright (C) 1998, 1999, 2000, 2001, 2002, 2003, 2004, 2005, 2006,
-   2007 Free Software Foundation, Inc.
+   2007, 2008 Free Software Foundation, Inc.
  
  This file is part of GNU Wget.
  
@@ -17,15 +17,16 @@ GNU General Public License for more details.
  You should have received a copy of the GNU General Public License
  along with Wget.  If not, see <http://www.gnu.org/licenses/>.
  
-In addition, as a special exception, the Free Software Foundation
-gives permission to link the code of its release of Wget with the
-OpenSSL project's "OpenSSL" library (or with modified versions of it
-that use the same license as the "OpenSSL" library), and distribute
-the linked executables.  You must obey the GNU General Public License
-in all respects for all of the code used other than "OpenSSL".  If you
-modify this file, you may extend this exception to your version of the
-file, but you are not obligated to do so.  If you do not wish to do
-so, delete this exception statement from your version.  */
+Additional permission under GNU GPL version 3 section 7
+
+If you modify this program, or any covered work, by linking or
+combining it with the OpenSSL project's OpenSSL library (or a
+modified version of that library), containing parts covered by the
+terms of the OpenSSL or SSLeay licenses, the Free Software Foundation
+grants you additional permission to convey the resulting work.
+Corresponding Source for a non-source form of such a combination
+shall include the source code for the parts of OpenSSL used as well
+as that of the covered work.  */
  
  #include "wget.h"
  
@@ -41,6 +42,7 @@ so, delete this exception statement from your version.  */
  #include "hash.h"
  #include "convert.h"
  #include "recur.h"              /* declaration of get_urls_html */
+#include "iri.h"
  
  struct map_context;
  
@@ -184,7 +186,7 @@ init_interesting (void)
       matches the user's preferences as specified through --ignore-tags
       and --follow-tags.  */
  
-  int i;
+  size_t i;
    interesting_tags = make_nocase_string_hash_table (countof (known_tags));
  
    /* First, add all the tags we know hot to handle, mapped to their
@@ -354,7 +356,8 @@ append_url (const char *link_uri,
  static void
  tag_find_urls (int tagid, struct taginfo *tag, struct map_context *ctx)
  {
-  int i, attrind;
+  size_t i;
+  int attrind;
    int first = -1;
  
    for (i = 0; i < countof (tag_url_attributes); i++)
@@ -381,7 +384,7 @@ tag_find_urls (int tagid, struct taginfo *tag, struct map_context *ctx)
        /* Find whether TAG/ATTRIND is a combination that contains a
           URL. */
        char *link = tag->attrs[attrind].value;
-      const int size = countof (tag_url_attributes);
+      const size_t size = countof (tag_url_attributes);
  
        /* If you're cringing at the inefficiency of the nested loops,
           remember that they both iterate over a very small number of
@@ -532,6 +535,25 @@ tag_handle_meta (int tagid, struct taginfo *tag, struct map_context *ctx)
            entry->link_expect_html = 1;
          }
      }
+  else if (http_equiv && 0 == strcasecmp (http_equiv, "content-type"))
+    {
+      /* Handle stuff like:
+         <meta http-equiv="Content-Type" content="text/html; charset=CHARSET"> */
+
+      char *mcharset;
+      char *content = find_attr (tag, "content", NULL);
+      if (!content)
+        return;
+
+      mcharset = parse_charset (content);
+      if (!mcharset)
+        return;
+
+      logprintf (LOG_VERBOSE, "Meta tag charset : %s\n", quote (mcharset));
+
+      /* sXXXav: Not used yet */
+      xfree (mcharset);
+    }
    else if (name && 0 == strcasecmp (name, "robots"))
      {
        /* Handle stuff like: