/* HTML parser for Wget.
Copyright (C) 1998, 2000 Free Software Foundation, Inc.
-This file is part of Wget.
+This file is part of GNU Wget.
-This program is free software; you can redistribute it and/or modify
+GNU Wget is free software; you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation; either version 2 of the License, or (at
your option) any later version.
-This program is distributed in the hope that it will be useful,
+GNU Wget is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
-along with this program; if not, write to the Free Software
+along with Wget; if not, write to the Free Software
Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA. */
/* The only entry point to this module is map_html_tags(), which see. */
as a backend for one.
Due to time and other constraints, this parser was not integrated
- into Wget until the version ???. */
+ into Wget until the version 1.7. */
/* DESCRIPTION:
#include <stdio.h>
#include <stdlib.h>
-#include <ctype.h>
#ifdef HAVE_STRING_H
# include <string.h>
#else
state = AC_S_DEFAULT;
break;
case AC_S_QUOTE1:
- assert (ch == '\'' || ch == '\"');
+ assert (ch == '\'' || ch == '"');
quote_char = ch; /* cheating -- I really don't feel like
introducing more different states for
different quote characters. */
SKIP_WS (p);
+ if (*p == '/')
+ {
+ /* A slash at this point means the tag is about to be
+ closed. This is legal in XML and has been popularized
+ in HTML via XHTML. */
+ /* <foo a=b c=d /> */
+ /* ^ */
+ ADVANCE (p);
+ SKIP_WS (p);
+ if (*p != '>')
+ goto backout_tag;
+ }
+
/* Check for end of tag definition. */
if (*p == '>')
break;
/* Establish bounds of attribute value. */
SKIP_WS (p);
- if (NAME_CHAR_P (*p) || *p == '>')
+ if (NAME_CHAR_P (*p) || *p == '/' || *p == '>')
{
/* Minimized attribute syntax allows `=' to be omitted.
For example, <UL COMPACT> is a valid shorthand for <UL
/* We skipped the whitespace and found something that is
neither `=' nor the beginning of the next attribute's
name. Back out. */
- goto backout_tag; /* <foo bar /... */
+ goto backout_tag; /* <foo bar [... */
/* ^ */
}