#include <config.h>
+#ifdef STANDALONE
+# define I_REALLY_WANT_CTYPE_MACROS
+#endif
+
#include <stdio.h>
#include <stdlib.h>
#ifdef HAVE_STRING_H
# define xmalloc malloc
# define xrealloc realloc
# define xfree free
+
+# define ISSPACE(x) isspace (x)
+# define ISDIGIT(x) isdigit (x)
+# define ISALPHA(x) isalpha (x)
+# define ISALNUM(x) isalnum (x)
+# define TOLOWER(x) tolower (x)
#endif /* STANDALONE */
-/* Pool support. For efficiency, map_html_tags() stores temporary
- string data to a single stack-allocated pool. If the pool proves
- too small, additional memory is allocated/resized with
- malloc()/realloc(). */
+/* Pool support. A pool is a resizable chunk of memory. It is first
+ allocated on the stack, and moved to the heap if it needs to be
+ larger than originally expected. map_html_tags() uses it to store
+ the zero-terminated names and values of tags and attributes.
+
+ Thus taginfo->name, and attr->name and attr->value for each
+ attribute, do not point into separately allocated areas, but into
+ different parts of the pool, separated only by terminating zeros.
+ This ensures minimum amount of allocation and, for most tags, no
+ allocation because the entire pool is kept on the stack. */
struct pool {
char *contents; /* pointer to the contents. */
state = AC_S_DEFAULT;
break;
case AC_S_QUOTE1:
- assert (ch == '\'' || ch == '"');
+ /* We must use 0x22 because broken assert macros choke on
+ '"' and '\"'. */
+ assert (ch == '\'' || ch == 0x22);
quote_char = ch; /* cheating -- I really don't feel like
introducing more different states for
different quote characters. */
SKIP_WS (p);
+ if (*p == '/')
+ {
+ /* A slash at this point means the tag is about to be
+ closed. This is legal in XML and has been popularized
+ in HTML via XHTML. */
+ /* <foo a=b c=d /> */
+ /* ^ */
+ ADVANCE (p);
+ SKIP_WS (p);
+ if (*p != '>')
+ goto backout_tag;
+ }
+
/* Check for end of tag definition. */
if (*p == '>')
break;
/* Establish bounds of attribute value. */
SKIP_WS (p);
- if (NAME_CHAR_P (*p) || *p == '>')
+ if (NAME_CHAR_P (*p) || *p == '/' || *p == '>')
{
/* Minimized attribute syntax allows `=' to be omitted.
For example, <UL COMPACT> is a valid shorthand for <UL
/* We skipped the whitespace and found something that is
neither `=' nor the beginning of the next attribute's
name. Back out. */
- goto backout_tag; /* <foo bar /... */
+ goto backout_tag; /* <foo bar [... */
/* ^ */
}