[svn] Squash embedded newlines in links.

[wget] / src / html-url.c
diff --git a/src/html-url.c b/src/html-url.c

index efbae629e8855c03322d2a422fe3e0cecab73edf..59d873b394e6443cbaacad970399fbf6b67596f3 100644 (file)
--- a/src/html-url.c
+++ b/src/html-url.c
@@ -81,6 +81,7 @@ enum {
    TAG_LAYER,
    TAG_LINK,
    TAG_META,
+  TAG_OBJECT,
    TAG_OVERLAY,
    TAG_SCRIPT,
    TAG_TABLE,
@@ -111,6 +112,7 @@ static struct known_tag {
    { TAG_LAYER,  "layer",       tag_find_urls },
    { TAG_LINK,   "link",        tag_handle_link },
    { TAG_META,   "meta",        tag_handle_meta },
+  { TAG_OBJECT,  "object",     tag_find_urls },
    { TAG_OVERLAY, "overlay",    tag_find_urls },
    { TAG_SCRIPT,         "script",      tag_find_urls },
    { TAG_TABLE,  "table",       tag_find_urls },
@@ -157,6 +159,7 @@ static struct {
    { TAG_IMG,           "src",          ATTR_INLINE },
    { TAG_INPUT,         "src",          ATTR_INLINE },
    { TAG_LAYER,         "src",          ATTR_INLINE | ATTR_HTML },
+  { TAG_OBJECT,                "data",         ATTR_INLINE },
    { TAG_OVERLAY,       "src",          ATTR_INLINE | ATTR_HTML },
    { TAG_SCRIPT,                "src",          ATTR_INLINE },
    { TAG_TABLE,         "background",   ATTR_INLINE },
@@ -227,9 +230,10 @@ init_interesting (void)
    /* Add the attributes we care about. */
    interesting_attributes = make_nocase_string_hash_table (10);
    for (i = 0; i < countof (additional_attributes); i++)
-    string_set_add (interesting_attributes, additional_attributes[i]);
+    hash_table_put (interesting_attributes, additional_attributes[i], "1");
    for (i = 0; i < countof (tag_url_attributes); i++)
-    string_set_add (interesting_attributes, tag_url_attributes[i].attr_name);
+    hash_table_put (interesting_attributes,
+                   tag_url_attributes[i].attr_name, "1");
  }
  
  /* Find the value of attribute named NAME in the taginfo TAG.  If the
@@ -328,7 +332,6 @@ append_url (const char *link_uri,
    DEBUGP (("appending \"%s\" to urlpos.\n", url->url));
  
    newel = xnew0 (struct urlpos);
-  newel->next = NULL;
    newel->url = url;
    newel->pos = tag->attrs[attrind].value_raw_beginning - ctx->text;
    newel->size = tag->attrs[attrind].value_raw_size;
@@ -609,9 +612,12 @@ get_urls_html (const char *file, const char *url, int *meta_disallow_follow)
      init_interesting ();
  
    /* Specify MHT_TRIM_VALUES because of buggy HTML generators that
-     generate <a href=" foo"> instead of <a href="foo"> (Netscape
-     ignores spaces as well.)  If you really mean space, use &32; or
-     %20.  */
+     generate <a href=" foo"> instead of <a href="foo"> (browsers
+     ignore spaces as well.)  If you really mean space, use &32; or
+     %20.  MHT_TRIM_VALUES also causes squashing of embedded newlines,
+     e.g. in <img src="foo.[newline]html">.  Such newlines are also
+     ignored by IE and Mozilla and are presumably introduced by
+     writing HTML with editors that force word wrap.  */
    flags = MHT_TRIM_VALUES;
    if (opt.strict_comments)
      flags |= MHT_STRICT_COMMENTS;
@@ -715,6 +721,10 @@ get_urls_file (const char *file)
  void
  cleanup_html_url (void)
  {
-  xfree_null (interesting_tags);
-  xfree_null (interesting_attributes);
+  /* Destroy the hash tables.  The hash table keys and values are not
+     allocated by this code, so we don't need to free them here.  */
+  if (interesting_tags)
+    hash_table_destroy (interesting_tags);
+  if (interesting_attributes)
+    hash_table_destroy (interesting_attributes);
  }