Release 1.6.0.

git-svn-id: http://htmlpurifier.org/svnroot/htmlpurifier/trunk@930 48356398-32a2-884e-a903-53898d9a118a
Add partial French install file.
2025-08-03 04:37:39 +02:00 · 2007-04-01 22:31:16 +00:00 · 2007-04-01 21:38:10 +00:00 · 2007-04-01 18:21:43 +00:00 · 2007-03-31 03:41:22 +00:00 · 2007-03-31 03:25:10 +00:00
195 changed files with 7491 additions and 1992 deletions
--- a/2
+++ b/2
@@ -4,7 +4,7 @@
 # Project related configuration options
 #---------------------------------------------------------------------------
 PROJECT_NAME           = HTML Purifier
-PROJECT_NUMBER         = 1.3.2
+PROJECT_NUMBER         = 1.6.0
 OUTPUT_DIRECTORY       = "C:/Documents and Settings/Edward/My Documents/My Webs/htmlpurifier/docs/doxygen"
 CREATE_SUBDIRS         = NO
 OUTPUT_LANGUAGE        = English
--- a/5
+++ b/5
@@ -8,6 +8,7 @@ installation GUI, you've come to the wrong place!)  The impatient can scroll
 down to the bottom of this INSTALL document to see the code, but you really
 should make sure a few things are properly done.

+Todo: Convert to using the array syntax for configuration.


 1.  Compatibility
@@ -46,7 +47,9 @@ HTML Purifier is all about web-standards, so accordingly your webpages should
 be standards compliant.  HTML Purifier can deal with these doctypes:

 * XHTML 1.0 Transitional (default)
+* XHTML 1.0 Strict
 * HTML 4.01 Transitional
+* HTML 4.01 Strict

 ...and these character encodings:

@@ -86,7 +89,7 @@ into configuring things just for the heck of it, skip to 4.3).
 * Am I using UTF-8?
 * Am I using XHTML 1.0 Transitional?

-If you answered yes to any of these questions, instantiate a configuration
+If you answered no to any of these questions, instantiate a configuration
 object and read on:

    $config = HTMLPurifier_Config::createDefault();
--- a/INSTALL.fr.utf8
+++ b/INSTALL.fr.utf8
@@ -0,0 +1,71 @@
+
+Installation
+    Comment installer HTML Purifier
+
+Attention: Ce document a encode en UTF-8. Si les lettres avec les accents
+est essoreuse, prenez un mieux editeur de texte.
+
+À L'Aide: Je ne suis pas un diseur natif de français. Si vous trouvez une
+erreur dans ce document, racontez-moi! Merci.
+
+
+L'installation de HTML Purifier est trés simple, parce qu'il ne doit pas
+la configuration.  Dans le pied de de document, les utilisateurs
+impatient peuvent trouver le code, mais je recommande que vous lisez
+ce document pour quelques choses.
+
+
+1.  Compatibilité
+
+HTML Purifier fonctionne dans PHP 4 et PHP 5. PHP 4.3.9 est le dernier
+version que je le testais.  Il ne dépend de les autre librairies.
+
+Les extensions optionnel est iconv (en général déjà installer) et 
+tidy (répandu aussi). Si vous utilisez UTF-8 et ne voulez pas
+l'indentation, vous pouvez utiliser HTML Purifier sans ces extensions.
+
+
+2.  Inclure la librarie
+
+Utilisez:
+
+    require_once '/path/to/library/HTMLPurifier.auto.php';
+
+...quand vous devez utiliser HTML Purifier (ne inclure pas quand vous
+ne devez pas, parce que HTML Purifier est trés grand.)
+
+Si vous n'aime pas que HTML Purifier change vos include_path, on peut
+change vos include_path, et:
+
+    require_once 'HTMLPurifier.php';
+
+Seuleument les contents dans library/ est essentiel; vous peut enlever
+les autre fichiers quand vous est dans une atmosphère professionnel.
+
+
+[En cours de construction]
+
+
+6.   Installation vite
+
+Si votre site web est en UTF-8 et XHTML Transitional, utilisez:
+
+<?php
+    require_once '/path/to/htmlpurifier/library/HTMLPurifier.auto.php';
+    
+    $purificateur = new HTMLPurifier();
+    $html_propre = $purificateur->purify($html_salle);
+?>
+
+Sinon, utilisez:
+
+<?php
+    require_once '/path/to/htmlpurifier/library/HTMLPurifier.auto.php';
+    
+    $config = HTMLPurifier_Config::createDefault();
+    $config->set('Core', 'Encoding', 'ISO-8859-1'); //remplacez avec votre encoding
+    $config->set('Core', 'XHTML', true); //remplacez avec false si HTML 4.01
+    $purificateur = new HTMLPurifier($config);
+    
+    $html_propre = $purificateur->purify($html_salle);
+?>
--- a/75
+++ b/75
@@ -9,11 +9,78 @@ NEWS ( CHANGELOG and HISTORY )                                     HTMLPurifier
    . Internal change
 ==========================

-1.4.0, unknown release date
-(major feature release)
+1.6.0, released 2007-04-01
+! Support for most common deprecated attributes via transformations:
+  + bgcolor in td, th, tr and table
+  + border in img
+  + name in a and img
+  + width in td, th and hr
+  + height in td, th
+! Support for CSS attribute 'height' added
+! Support for rel and rev attributes in a tags added, use %Attr.AllowedRel
+  and %Attr.AllowedRev to activate
+- You can define ID blacklists using regular expressions via
+  %Attr.IDBlacklistRegexp
+- Error messages are emitted when you attempt to "allow" elements or
+  attributes that HTML Purifier does not support

-1.3.3, unknown release date, may be dropped
-(security/bugfix/minor feature release)
+1.5.1, unknown release date
+- Fix segfault in unit test. The problem is not very reproduceable and
+  I don't know what causes it, but a six line patch fixed it.
+
+1.5.0, released 2007-03-23
+! Added a rudimentary I18N and L10N system modeled off MediaWiki. It
+  doesn't actually do anything yet, but keep your eyes peeled.
+! docs/enduser-utf8.html explains how to use UTF-8 and HTML Purifier
+! Newly structured HTMLDefinition modeled off of XHTML 1.1 modules.
+  I am loathe to release beta quality APIs, but this is exactly that;
+  don't use the internal interfaces if you're not willing to do migration
+  later on.
+- Allow 'x' subtag in language codes
+- Fixed buggy chameleon-support for ins and del
+. Added support for IDREF attributes (i.e. for)
+. Renamed HTMLPurifier_AttrDef_Class to HTMLPurifier_AttrDef_Nmtokens
+. Removed context variable ParentType, replaced with IsInline, which
+  is false when you're not inline and an integer of the parent that
+  caused you to become inline when you are (so possibly zero)
+. Removed ElementDef->type in favor of ElementDef->descendants_are_inline
+  and HTMLDefinition->content_sets
+. StrictBlockquote now reports what elements its supposed to allow,
+  rather than what it does allow
+. Removed HTMLDefinition->info_flow_elements in favor of
+  HTMLDefinition->content_sets['Flow']
+. Removed redundant "exclusionary" definitions from DTD roster
+. StrictBlockquote now requires a construction parameter as if it
+  were an Required ChildDef, this is the "real" set of allowed elements
+. AttrDef partitioned into HTML, CSS and URI segments
+. Modify Youtube filter regexp to be multiline
+. Require both PHP5 and DOM extension in order to use DOMLex, fixes
+  some edge cases where a DOMDocument class exists in a PHP4 environment
+  due to DOM XML extension.
+
+1.4.1, released 2007-01-21
+! docs/enduser-youtube.html updated according to new functionality
+- YouTube IDs can have underscores and dashes
+
+1.4.0, released 2007-01-21
+! Implemented list-style-image, URIs now allowed in list-style
+! Implemented background-image, background-repeat, background-attachment
+  and background-position CSS properties. Shorthand property background
+  supports all of these properties.
+! Configuration documentation looks nicer
+! Added %Core.EscapeNonASCIICharacters to workaround loss of Unicode
+  characters while %Core.Encoding is set to a non-UTF-8 encoding.
+! Support for configuration directive aliases added
+! Config object can now be instantiated from ini files
+! YouTube preservation code added to the core, with two lines of code
+  you can add it as a filter to your code. See smoketests/preserveYouTube.php
+  for sample code.
+! Moved SLOW to docs/enduser-slow.html and added code examples
+- Replaced version check with functionality check for DOM (thanks Stephen
+  Khoo)
+. Added smoketest 'all.php', which loads all other smoketests via frames
+. Implemented AttrDef_CSSURI for url(http://google.com) style declarations
+. Added convenient single test selector form on test runner

 1.3.2, released 2006-12-25
 ! HTMLPurifier object now accepts configuration arrays, no need to manually
--- a/25
+++ b/25
@@ -1,13 +1,22 @@

 README
-    All about HTMLPurifier
+    All about HTML Purifier

-HTMLPurifier is an HTML filtering solution.  It uses a unique combination of
-robust whitelists and agressive parsing to ensure that not only are XSS
-attacks thwarted, but the resulting HTML is standards compliant.
+HTML Purifier is an HTML filtering solution that uses a unique combination 
+of robust whitelists and agressive parsing to ensure that not only are 
+XSS attacks thwarted, but the resulting HTML is standards compliant. 

-See INSTALL on how to use the library.  See docs/ for more developer-oriented
-documentation as well as some code examples.  Users of TinyMCE or FCKeditor
-may be especially interested in WYSIWYG.
+HTML Purifier is oriented towards richly formatted documents from 
+untrusted sources that require CSS and a full tag-set.  This library can 
+be configured to accept a more restrictive set of tags, but it won't be 
+as efficient as more bare-bones parsers. It will, however, do the job 
+right, which may be more important. 

-HTMLPurifier can be found on the web at: http://hp.jpsband.org/
+Places to go:
+
+* See INSTALL for a quick installation guide
+* See docs/ for developer-oriented documentation, code examples and
+  an in-depth installation guide.
+* See WYSIWYG for information on editors like TinyMCE and FCKeditor
+
+HTML Purifier can be found on the web at: http://hp.jpsband.org/
--- a/40
+++ b/40
@@ -1,40 +0,0 @@
-
-SLOW
-  also known as the HELP ME LIBRARY IS TOO SLOW MY PAGE TAKE TOO LONG LOAD page
-
-HTML Purifier is a very powerful library.  But with power comes great
-responsibility, or, at least, longer execution times.  Remember, this
-library isn't lightly grazing over submitted HTML: it's deconstructing
-the whole thing, rigorously checking the parts, and then putting it
-back together.
-
-So, if it so turns out that HTML Purifier is kinda too slow for outbound
-filtering, you've got a few options:
-
-1. Inbound filtering - perform filtering of HTML when it's submitted by the
-user.  Since the user is already submitting something, an extra half a
-second tacked on to the load time probably isn't going to be that huge of
-a problem.  Then, displaying the content is a simple a manner of outputting
-it directly from your database/filesystem.  The trouble with this method is
-that your user loses the original text, and when doing edits, will be
-handling the filtered text.  While this may be a good thing, especially if
-you're using a WYSIWYG editor, it can also result in data-loss if a user
-makes a typo.
-
-2. Caching the filtered output - accept the submitted text and put it
-unaltered into the database, but then also generate a filtered version and
-stash that in the database.  Serve the filtered version to readers, and the
-unaltered version to editors.  If need be, you can invalidate the cache and
-have the cached filtered version be regenerated on the first page view.  Pros?
-Full data retention.  Cons?  It's more complicated, and opens other editors
-up to XSS if they are using a WYSIWYG editor (to fix that, they'd have to
-be able to get their hands on the *really* original text served in plaintext
-mode).
-
-In short, inbound filtering is almost as simple as outbound filtering, but
-it has some drawbacks which cannot be fixed unless you save both the original
-and the filtered versions.
-
-There is a third option: profile and optimize HTMLPurifier yourself.  Be sure
-to report back your results if you decide to do that!  Especially if you
-port HTML Purifier to C++.  ;-)
--- a/90
+++ b/90
@@ -4,46 +4,67 @@ TODO List
 = KEY ====================
    # Flagship
    - Regular
-    ? At-risk
+    ? Maybe I'll Do It
 ==========================

-1.4 release
- # More extensive URI filtering schemes (see docs/proposal-new-directives.txt)
- # Allow for background-image and list-style-image (intrinsically tied to above)
- # Add hooks for custom behavior (for instance, YouTube preservation)
- - Aggressive caching
- ? Rich set* methods and config file loaders for HTMLPurifier_Config
- ? Configuration profiles: sets of directives that get set with one func call
- ? ConfigSchema directive aliases (so we can rename some of them)
- ? URI validation routines tighter (see docs/dev-code-quality.html) (COMPLEX)
+1.7 release [Advanced API]
+ # Complete advanced API, and fully document it
+ # Implement all edge-case attribute transforms
+ # Implement all deprecated tags and attributes
+ - Parse TinyMCE-style whitelist into our %HTML.Allow* whitelists (possibly
+   do this earlier)

-1.5 release
+1.8 release [Refactor, refactor!]
+ # URI validation routines tighter (see docs/dev-code-quality.html) (COMPLEX)
+ # Advanced URI filtering schemes (see docs/proposal-new-directives.txt)
+ - Configuration profiles: predefined directives set with one func call
+ - Implement IDREF support (harder than it seems, since you cannot have
+   IDREFs to non-existent IDs)
+ - Allow non-ASCII characters in font names
+
+1.9 release [Error'ed]
 # Error logging for filtering/cleanup procedures
    - Requires I18N facilities to be created first (COMPLEX)
-
-1.6 release
- # Add pre-packaged "levels" of cleaning (custom behavior already done)
+ - XSS-attempt detection
 - More fine-grained control over escaping behavior
    - Silently drop content inbetween SCRIPT tags (can be generalized to allow
      specification of elements that, when detected as foreign, trigger removal
      of children, although unbalanced tags could wreck havoc (or at least
      delete the rest of the document)).

-1.7 release
+1.10 release [Do What I Mean, Not What I Say]
 # Additional support for poorly written HTML
-    - Implement all non-essential attribute transforms (BIG!)
    - Microsoft Word HTML cleaning (i.e. MsoNormal, but research essential!)
    - Friendly strict handling of <address> (block -> <br>)
+ - Remove redundant tags, ex. <u><u>Underlined</u></u>. Implementation notes:
+    1. Analyzing which tags to remove duplicants
+    2. Ensure attributes are merged into the parent tag
+    3. Extend the tag exclusion system to specify whether or not the
+    contents should be dropped or not (currently, there's code that could do
+    something like this if it didn't drop the inner text too.)
+ - Remove <span> tags that don't do anything (no attributes)
+ - Remove empty inline tags<i></i>
+ - Append something to duplicate IDs so they're still usable (impl. note: the
+   dupe detector would also need to detect the suffix as well)

-2.0 release
+2.0 release [Beyond HTML]
+ # Legit token based CSS parsing (will require revamping almost every
+   AttrDef class)
 # Formatters for plaintext (COMPLEX)
    - Auto-paragraphing (be sure to leverage fact that we know when things
      shouldn't be paragraphed, such as lists and tables).
    - Linkify URLs
    - Smileys
    - Linkification for HTML Purifier docs: notably configuration and classes
+ - Allow tags to be "armored", an internal flag that protects them
+   from validation and passes them out unharmed
+ - Fixes for Firefox's inability to handle COL alignment props (Bug 915)
+ - Automatically add non-breaking spaces to empty table cells when
+   empty-cells:show is applied to have compatibility with Internet Explorer
+ - Convert RTL/LTR override characters to <bdo> tags, or vice versa on demand.
+   Also, enable disabling of directionality

-3.0 release
+3.0 release [To XML and Beyond]
 - Extended HTML capabilities based on namespacing and tag transforms (COMPLEX)
    - Hooks for adding custom processors to custom namespaced tags and
      attributes, offer default implementation
@@ -53,43 +74,18 @@ TODO List
 Ongoing
 - Lots of profiling, make it faster!
 - Plugins for major CMSes (COMPLEX)
-    - Drupal
-    - WordPress
+    - WordPress (mostly written, needs beta-testing)
    - eFiction
    - more! (look for ones that use WYSIWYGs)

 Unknown release (on a scratch-an-itch basis)
- - Fixes for Firefox's inability to handle COL alignment props (Bug 915)
- - Automatically add non-breaking spaces to empty table cells when
-   empty-cells:show is applied to have compatibility with Internet Explorer
- - Convert RTL/LTR override characters to <bdo> tags, or vice versa on demand.
-   Also, enable disabling of directionality
- - Append something to duplicate IDs so they're still usable (impl. note: the
-   dupe detector would also need to detect the suffix as well)
- - Have 'lang' attribute be checked against official lists
-
-Encoding workarounds
- - Non-lossy dumb alternate character encoding transformations, achieved by
-   numerically encoding all non-ASCII characters
- - Semi-lossy dumb alternate character encoding transformations, achieved by
+ ? Semi-lossy dumb alternate character encoding transfor
+ ? Have 'lang' attribute be checked against official lists, achieved by
   encoding all characters that have string entity equivalents

 Requested
- - Native content compression, whitespace stripping (don't rely on Tidy, make
+ ? Native content compression, whitespace stripping (don't rely on Tidy, make
   sure we don't remove from <pre> or related tags)
- - Win32 Phalanger C# binaries (?)
- - Remove redundant tags, ex. <u><u>Underlined</u></u>. Implementation notes:
-    1. Analyzing which tags to remove duplicants
-    2. Ensure attributes are merged into the parent tag
-    3. Extend the tag exclusion system to specify whether or not the
-    contents should be dropped or not (currently, there's code that could do
-    something like this if it didn't drop the inner text too.)
- - More user-friendly warnings when %HTML.Allow* attempts to specify a
-   tag or attribute that is not supported
- - Allow specifying global attributes on a tag-by-tag basis in
-   %HTML.AllowAttributes
- - Parse TinyMCE whitelist into our %HTML.Allow* whitelists
- - XSS-attempt detection

 Wontfix
 - Non-lossy smart alternate character encoding transformations (unless
--- a/3
+++ b/3
@@ -18,4 +18,5 @@ HTML Purifier is perfect for filtering pure-HTML input from WYSIWYG editors.
 Enough said.

 There is a proof-of-concept integration of HTML Purifier with the Mantis
-bugtracker at http://hp.jpsband.org/mantis/
+bugtracker at http://hp.jpsband.org/mantis/ You can see notes on how
+this integration was acheived at http://hp.jpsband.org/mantis_notes.txt
--- a/art/1000passes.png
+++ b/art/1000passes.png
--- a/benchmarks/Lexer.php
+++ b/benchmarks/Lexer.php
@@ -7,6 +7,7 @@ set_include_path(get_include_path() . PATH_SEPARATOR . '../library/');

 require_once 'HTMLPurifier/ConfigSchema.php';
 require_once 'HTMLPurifier/Config.php';
+require_once 'HTMLPurifier/Context.php';

 $LEXERS = array();
 $RUNS = isset($GLOBALS['HTMLPurifierTest']['Runs'])
@@ -93,11 +94,14 @@ function print_lexers() {
 function do_benchmark($name, $document) {
    global $LEXERS, $RUNS;
    
+    $config = HTMLPurifier_Config::createDefault();
+    $context = new HTMLPurifier_Context();
+    
    $timer = new RowTimer($name);
    $timer->start();
    
    foreach($LEXERS as $key => $lexer) {
-        for ($i=0; $i<$RUNS; $i++) $tokens = $lexer->tokenizeHTML($document);
+        for ($i=0; $i<$RUNS; $i++) $tokens = $lexer->tokenizeHTML($document, $config, $context);
        $timer->setMarker($key);
    }
    
--- a/benchmarks/ProfileDirectLex.php
+++ b/benchmarks/ProfileDirectLex.php
@@ -5,12 +5,15 @@ set_include_path(get_include_path() . PATH_SEPARATOR . '../library/');
 require_once 'HTMLPurifier/ConfigSchema.php';
 require_once 'HTMLPurifier/Config.php';
 require_once 'HTMLPurifier/Lexer/DirectLex.php';
+require_once 'HTMLPurifier/Context.php';

 $input = file_get_contents('samples/Lexer/4.html');
 $lexer = new HTMLPurifier_Lexer_DirectLex();
+$config = HTMLPurifier_Config::createDefault();
+$context = new HTMLPurifier_Context();

 for ($i = 0; $i < 10; $i++) {
-    $tokens = $lexer->tokenizeHTML($input);
+    $tokens = $lexer->tokenizeHTML($input, $config, $context);
 }

 ?>
--- a/configdoc/generate.php
+++ b/configdoc/generate.php
@@ -99,6 +99,8 @@ foreach($schema->info as $namespace_name => $namespace_info) {
    
    foreach ($namespace_info as $name => $info) {
        
+        if ($info->class == 'alias') continue;
+        
        $dom_directive = $dom_document->createElement('directive');
        $dom_namespace->appendChild($dom_directive);
        
@@ -186,7 +188,7 @@ $xsl_processor->importStylesheet($xsl_dom_stylesheet);
 $html_output = $xsl_processor->transformToXML($dom_document);

 // some slight fudges to preserve backwards compatibility
-$html_output = str_replace('/>', ' />', $html_output); // <br /> not <br>
+$html_output = str_replace('/>', ' />', $html_output); // <br /> not <br/>
 $html_output = str_replace(' xmlns=""', '', $html_output); // rm unnecessary xmlns

 if (class_exists('Tidy')) {
--- a/configdoc/styles/plain.css
+++ b/configdoc/styles/plain.css
@@ -1,3 +1,6 @@
+
+body {margin:1em 4em;}
+
 table {border-collapse:collapse;}
 table td, table th {padding:0.2em;}

@@ -8,3 +11,14 @@ table.constraints td pre {margin:0;}

 #toc {list-style-type:none; font-weight:bold;}
 #toc ul {list-style-type:disc; font-weight:normal;}
+
+.description p {margin-top:0;margin-bottom:1em;}
+
+#library, h1 {text-align:center; font-family:Garamond, serif;
+  font-variant:small-caps;}
+#library {font-size:1em;}
+h1 {margin-top:0;}
+h2 {border-bottom:1px solid #CCC; font-family:sans-serif; font-weight:normal;
+    font-size:1.3em;}
+h3 {font-family:sans-serif; font-size:1.1em; font-weight:bold; }
+h4 {font-family:sans-serif; font-size:0.9em; font-weight:bold; }
--- a/configdoc/styles/plain.xsl
+++ b/configdoc/styles/plain.xsl
@@ -18,12 +18,13 @@
    <xsl:template match="/">
        <html lang="en" xml:lang="en">
            <head>
-                <title><xsl:value-of select="/configdoc/title" /> Configuration Documentation</title>
+                <title>Configuration Documentation - <xsl:value-of select="/configdoc/title" /></title>
                <meta http-equiv="Content-Type" content="text/html;charset=UTF-8" />
                <link rel="stylesheet" type="text/css" href="styles/plain.css" />
            </head>
            <body>
-                <h1><xsl:value-of select="/configdoc/title" /> Configuration Documentation</h1>
+                <div id="library"><xsl:value-of select="/configdoc/title" /></div>
+                <h1>Configuration Documentation</h1>
                <h2>Table of Contents</h2>
                <ul id="toc">
                    <xsl:apply-templates mode="toc" />
--- a/docs/dev-advanced-api.html
+++ b/docs/dev-advanced-api.html
@@ -0,0 +1,287 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Strict//EN"
+    "http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd">
+<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en"><head>
+<meta http-equiv="Content-Type" content="text/html; charset=UTF-8" />
+<meta name="description" content="Functional specification for HTML Purifier's advanced API for defining custom filtering behavior." />
+<link rel="stylesheet" type="text/css" href="style.css" />
+
+<title>Advanced API - HTML Purifier</title>
+
+</head><body>
+
+<h1>Advanced API</h1>
+
+<div id="filing">Filed under Development</div>
+<div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>
+
+<p>HTML Purifier currently natively supports only a subset of HTML's
+allowed elements, attributes, and behavior. This is by design,
+but as the user is always right, they'll need some method to overload
+these behaviors.</p>
+
+<p>Our goals are to let the user:</p>
+
+<dl>
+    <dt>Select</dt>
+    <dd><ul>
+        <li>Doctype</li>
+        <li>Mode: Lenient / Correctional</li>
+        <li>Elements / Attributes / Modules</li>
+        <li>Filterset</li>
+    </ul></dd>
+    <dt>Customize</dt>
+    <dd><ul>
+        <li>Attributes</li>
+        <li>Elements</li>
+    </ul></dd>
+    <dt>Internals</dt>
+    <dd><ul>
+        <li>Modules / Elements / Attributes / Attribute Types</li>
+        <li>Filtersets</li>
+        <li>Doctype</li>
+    </ul></dd>
+</dl>
+
+<h2>Select</h2>
+
+<p>For basic use, the user will have to specify some basic parameters. This
+is not strictly necessary, as HTML Purifier's default setting will always
+output safe code, but is required for standards-compliant output.</p>
+
+<h3>Selecting a Doctype</h3>
+
+<p>The first thing to select is the <strong>doctype</strong>. This
+is essential for standards-compliant output.</p>
+
+<p class="technical">This identifier is based
+on the name the W3C has given to the document type and <em>not</em>
+the DTD identifier.</p>
+
+<p>This parameter is set via the configuration object:</p>
+
+<pre>$config->set('HTML', 'Doctype', 'XHTML 1.0 Transitional');</pre>
+
+<p>Due to historical reasons, the default doctype is XHTML 1.0
+Transitional, however, we really shouldn't be guessing what the user's
+doctype is. Fortunantely, people who can't be bothered to set this won't
+be bothered when their pages stop validating.</p>
+
+<h3>Selecting Mode</h3>
+
+<p>Within doctypes, there are various <strong>modes</strong> of operation.
+These indicate variant behaviors that, while not strictly changing the
+allowed set of elements and attributes, definitely affect the output.
+Currently, we have two modes, which may be used together:</p>
+
+<dl>
+    <dt>Lenient</dt>
+    <dd>
+        <p>Deprecated elements and attributes will be transformed into
+        standards-compliant alternatives when explicitly disallowed.</p>
+        <p>For example, in the XHTML 1.0 Strict doctype, a <code>center</code>
+        element would be turned into a <code>div</code> with the CSS property
+        <code>text-align:center;</code>, but in XHTML 1.0 Transitional
+        the element would be preserved.</p>
+        <p>This mode is on by default.</p>
+    </dd>
+    <dt>Correctional[items to correct]</dt>
+    <dd>
+        <p>Deprecated elements and attributes will be transformed into
+        standards-compliant alternatives whenever possible.
+        It may have various levels of operation.</p>
+        <p>Referring back to the previous example, the <code>center</code> element would
+        be transformed in both cases. However, elements without a
+        reasonable standards-compliant alternative will be preserved
+        in their form.</p>
+        <p>A user may want to correct certain deprecated attributes, but
+        not others. For example, the <code>bgcolor</code> attribute may be
+        acceptable, but the <code>center</code> element not; also, possibly,
+        an HTML Purifier transformation may be buggy, so the user wants
+        to forgo it. Thus, correctional accepts an array defining which
+        elements and attributes to cleanup, or no parameter at all, which
+        means everything gets corrected. This also means that each
+        correction needs to be given a unique ID that can be referenced
+        in this manner. (We may also allow globbing, like *.name or a.*
+        for mass-enabling correction, and subtractive mode, where things
+        specified stop correction.) This array gets passed into the
+        constructor of the mode's module.</p>
+        <p>This mode is on by default.</p>
+    </dd>
+</dl>
+
+<p>A possible call to select modes would be:</p>
+
+<pre>$config->set('HTML', 'Mode', array('correctional', 'lenient'));</pre>
+
+<p>If modes have extra parameters, a hash is necessary:</p>
+
+<pre>$config->set('HTML', 'Mode', array(
+    'correctional' => 'center,a.name',
+    'lenient' => true // this one's just boolean
+));</pre>
+
+<p>Modes may be specified along with the doctype declaration (we may want
+to get a better set of separator characters):</p>
+
+<pre>$config->setDoctype('XHTML Transitional 1.0', '+correctional[center,a.name] -lenient');</pre>
+
+<p>
+With regards to the various levels of operation conjectured in the
+Correctional mode, this is prompted by the fact that a user may want to
+correct certain problems but not others, for example, fix the <code>center</code>
+element but not the <code>u</code> element, both of which are deprecated.
+Having an integer <q>level</q> will not work very well for such fine
+grained tweaking, but an array of specific settings might.</p>
+
+<h3>Selecting Elements / Attributes / Modules</h3>
+
+<p></p>
+
+<p>If this cookie cutter approach doesn't appeal to a user, they may
+decide to roll their own filterset by selecting modules, elements and
+attributes to allow.</p>
+
+<p class="technical">This would make use of the same facilities
+as a filterset author would use, except that it would go under an
+<q>anonymous</q> filterset that would be auto-selected if any of the
+relevant module/elements/attribute selection configuration directives were
+non-null.</p>
+
+<p>In practice, this is the most commonly demanded feature. Most users are
+perfectly happy defining a filterset that looks like:</p>
+
+<pre>$config->setAllowedHTML('a[href,title];em;p;blockquote');</pre>
+
+<p class="technical">The directive %HTML.Allowed is a convenience function
+that may be fully expressed with the legacy interface, and thus is
+given its own setter.</p>
+
+<p>We currently support a separated interface, which also must be preserved:</p>
+
+<pre>$config->set('HTML', 'AllowedElements', 'a,em,p,blockquote');
+$config->set('HTML', 'AllowedAttributes', 'a.href,a.title');</pre>
+
+<p>A user may also choose to allow modules:</p>
+
+<pre>$config->set('HTML', 'AllowedModules', 'Hypertext,Text,Lists'); // or
+$config->setAllowedHTML('Hypertext,Text,Lists');</pre>
+
+<p>But it is not expected that this feature will be widely used.</p>
+
+<p class="fixme">The granularity of these modules is too coarse for
+the average user (for example, the core module loads everything from
+the essential <code>p</code> element to the not-so-safe <code>h1</code>
+element). How do we make this still a viable solution? Possible answers
+may be sub-modules or module parameters. This may not even be a problem,
+considering that most people won't be selecting modules.</p>
+
+<p class="technical">Modules are distinguished from regular elements by the
+case of their first letter. While XML distinguishes between and allows
+lower and uppercase letters in element names, most well-known XML
+languages use only lower-case
+element names for sake of consistency.</p>
+
+<p class="technical">Considering that, internally speaking, as mandated by
+the XHTML 1.1 Modularization specification, we have organized our
+elements around modules, considerable gymnastics will be needed to
+get this sort of functionality working.</p>
+
+<h3>Unified selector</h3>
+
+<p>Because selecting each and every one of these configuration options
+is a chore, we may wish to offer a specialized configuration method
+for selecting a filterset. Possibility:</p>
+
+<pre>function selectFilter($doctype, $filterset, $mode)</pre>
+
+<p>...which is simply a light wrapper over the individual configuration
+calls. A custom config file format or text format could also be adopted.</p>
+
+<h2>Customize</h2>
+
+<p>By reviewing topic posts in the support forum, we determined that
+there were two primarily demanded customization features people wanted:
+to add an attribute to an existing element, and to add an element.
+Thus, we'll want to create convenience functions for these common
+use-cases.</p>
+
+<p>Note that the functions described here are only available if
+a raw copy of <code>HTMLPurifier_HTMLDefinition</code> was retrieved.
+<code>addAttribute</code> may work on a processed copy, but for
+consistency's sake we will mandate this for everything.</p>
+
+<h3>Attributes</h3>
+
+<p>An attribute is bound to an element by a name and has a specific
+<code>AttrDef</code> that validates it. Thus, the interface should
+be:</p>
+
+<pre>function addAttribute($element, $attribute, $attribute_def);</pre>
+
+<p>With a use-case that looks like:</p>
+
+<pre>$def->addAttribute('a', 'rel', new HTMLPurifier_AttrDef_Enum(array('nofollow')));</pre>
+
+<p>The <code>$attribute_def</code> value can be a little flexible,
+to make things simpler. We'll let it also be:</p>
+
+<ul>
+    <li>Class name: We'll instantiate it for you</li>
+    <li>Function name: We'll create an <code>HTMLPurifier_AttrDef_Anonymous</code>
+        class with that function registered as a callback.</li>
+    <li>String attribute type: We'll use <code>HTMLPurifier_AttrTypes</code>
+        </li>
+    <li>String starting with <code>enum(</code>: We'll explode it and stuff it in an
+        <code>HTMLPurifier_AttrDef_Enum</code> for you.</li>
+</ul>
+
+<p>Making the previous example written as:</p>
+
+<pre>$def->addAttribute('a', 'rel', 'enum(nofollow)');</pre>
+
+<h3>Elements</h3>
+
+<p>An element requires certain information as specified by
+<code>HTMLPurifier_ElementDef</code>. However, not all of it is necessary,
+the usual things required are:</p>
+
+<ul>
+    <li>Attributes</li>
+    <li>Content model/type</li>
+    <li>Registration in a content set</li>
+</ul>
+
+<p>This suggests an API like this:</p>
+
+<pre>function addElement($element, $type, $content_model, $attributes = array());</pre>
+
+<p>Each parameter explained in depth:</p>
+
+<dl>
+    <dt><code>$element</code></dt>
+    <dd>Element name, ex. 'label'</dd>
+    <dt><code>$type</code></dt>
+    <dd>Content set to register in, ex. 'Inline' or 'Flow'</dd>
+    <dt><code>$content_model</code></dt>
+    <dd>Description of allowed children. This is a merged form of
+        <code>HTMLPurifier_ElementDef</code>'s member variables
+        <code>$content_model</code> and <code>$content_model_type</code>,
+        where the form is <q>Type: Model</q>, ex. 'Optional: Inline'.</dd>
+    <dt><code>$attributes</code></dt>
+    <dd>Array of attribute names to attribute definitions, much like
+        the above-described attribute customization.</dd>
+</dl>
+
+<p>A possible usage:</p>
+
+<pre>$def->addElement('font', 'Inline', 'Optional: Inline',
+    array(0 => array('Common'), 'color' => 'Color'));</pre>
+
+<p>We may want to Common attribute collection inclusion to be added
+by default.</p>
+
+<div id="version">$Id$</div>
+
+</body></html>
--- a/docs/dev-code-quality.html
+++ b/docs/dev-code-quality.html
@@ -14,6 +14,7 @@

 <div id="filing">Filed under Development</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Okay, face it.  Programmers can get lazy, cut corners, or make mistakes. They
 also can do quick prototypes, and then forget to rewrite them later.  Well,
--- a/docs/dev-naming.html
+++ b/docs/dev-naming.html
@@ -14,6 +14,7 @@

 <div id="filing">Filed under Development</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>The classes in this library follow a few naming conventions, which may
 help you find the correct functionality more quickly.  Here they are:</p>
--- a/docs/dev-optimization.html
+++ b/docs/dev-optimization.html
@@ -14,6 +14,7 @@

 <div id="filing">Filed under Development</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Here are some possible optimization techniques we can apply to code sections if
 they turn out to be slow.  Be sure not to prematurely optimize: if you get
--- a/docs/dev-progress.html
+++ b/docs/dev-progress.html
@@ -32,6 +32,7 @@ thead th {text-align:left;padding:0.1em;background-color:#EEE;}

 <div id="filing">Filed under Development</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <h2>Key</h2>

@@ -59,7 +60,7 @@ thead th {text-align:left;padding:0.1em;background-color:#EEE;}
 <tbody>
 <tr><th colspan="2">Standard</th></tr>
 <tr class="css1 impl-yes"><td>background-color</td><td>COMPOSITE(&lt;color&gt;, transparent)</td></tr>
-<tr class="css1 impl-yes"><td>background</td><td>SHORTHAND, only for color, see below for info on background-image and friends</td></tr>
+<tr class="css1 impl-yes"><td>background</td><td>SHORTHAND, currently alias for background-color</td></tr>
 <tr class="css1 impl-yes"><td>border</td><td>SHORTHAND, MULTIPLE</td></tr>
 <tr class="css1 impl-yes"><td>border-color</td><td>MULTIPLE</td></tr>
 <tr class="css1 impl-yes"><td>border-style</td><td>MULTIPLE</td></tr>
@@ -141,17 +142,17 @@ thead th {text-align:left;padding:0.1em;background-color:#EEE;}

 <tbody>
 <tr><th colspan="2">Unknown</th></tr>
-<tr class="danger css1"><td>background-image</td><td>Dangerous, target milestone 1.3</td></tr>
-<tr class="css1"><td>background-attachment</td><td>ENUM(scroll, fixed),
+<tr class="danger css1 impl-yes"><td>background-image</td><td>Dangerous, target milestone 1.3</td></tr>
+<tr class="css1 impl-yes"><td>background-attachment</td><td>ENUM(scroll, fixed),
    Depends on background-image</td></tr>
-<tr class="css1"><td>background-position</td><td>Depends on background-image</td></tr>
+<tr class="css1 impl-yes"><td>background-position</td><td>Depends on background-image</td></tr>
 <tr class="danger impl-no"><td>cursor</td><td>Dangerous but fluffy</td></tr>
 <tr class="danger css1"><td>display</td><td>ENUM(...), Dangerous but interesting;
    will not implement list-item, run-in (Opera only) or table (no IE);
    inline-block has incomplete IE6 support and requires -moz-inline-box
    for Mozilla. Unknown target milestone.</td></tr>
-<tr><td class="css1">height</td><td>Interesting, why use it? Unknown target milestone.</td></tr>
-<tr class="danger css1"><td>list-style-image</td><td>Dangerous? Target milestone 1.3</td></tr>
+<tr class="css1 impl-yes"><td>height</td><td>Interesting, why use it? Unknown target milestone.</td></tr>
+<tr class="danger css1 impl-yes"><td>list-style-image</td><td>Dangerous?</td></tr>
 <tr class="impl-no"><td>max-height</td><td rowspan="4">No IE 5/6</td></tr>
 <tr class="impl-no"><td>min-height</td></tr>
 <tr class="impl-no"><td>max-width</td></tr>
@@ -230,7 +231,7 @@ Mozilla on inside and needs -moz-outline, no IE support.</td></tr>

 <tbody>
 <tr><th colspan="3">CSS</th></tr>
-<tr class="impl-yes"><td>style</td><td>All</td><td>Not all properties may be implemented, parser is good though.</td></tr>
+<tr class="impl-yes"><td>style</td><td>All</td><td>Parser is reasonably functional. Status here doesn't count individual properties.</td></tr>
 </tbody>

 <tbody>
@@ -243,8 +244,8 @@ Mozilla on inside and needs -moz-outline, no IE support.</td></tr>
 <tbody>
 <tr><th colspan="3">Miscellaneous</th></tr>
 <tr><td>datetime</td><td>DEL, INS</td><td>No visible effect, ISO format</td></tr>
-<tr><td>rel</td><td>A</td><td>Largely user-defined: nofollow, tag (see microformats)</td></tr>
-<tr><td>rev</td><td>A</td><td>Largely user-defined: vote-*</td></tr>
+<tr class="impl-yes"><td>rel</td><td>A</td><td>Largely user-defined: nofollow, tag (see microformats)</td></tr>
+<tr class="impl-yes"><td>rev</td><td>A</td><td>Largely user-defined: vote-*</td></tr>
 <tr class="feature"><td>axis</td><td>TD, TH</td><td>W3C only: No browser implementation</td></tr>
 <tr class="feature"><td>char</td><td>COL, COLGROUP, TBODY, TD, TFOOT, TH, THEAD, TR</td><td>W3C only: No browser implementation</td></tr>
 <tr class="feature"><td>headers</td><td>TD, TH</td><td>W3C only: No browser implementation</td></tr>
@@ -261,28 +262,28 @@ Mozilla on inside and needs -moz-outline, no IE support.</td></tr>
 </tbody>

 <tbody>
-<tr><th colspan="3">Transform, target milestone 1.4</th></tr>
+<tr><th colspan="3">Transform, target milestone 1.6</th></tr>
 <tr><td rowspan="5">align</td><td>CAPTION</td><td>Near-equiv style 'caption-side', drop left and right</td></tr>
    <tr><td>IMG</td><td rowspan="2">Margin-left and margin-right = auto or parent div</td></tr>
    <tr><td>TABLE</td></tr>
-    <tr><td>HR</td><td>Equivalent style 'text-align' (IE tested)</td></tr>
+    <tr><td>HR</td><td>Near-equivalent style 'text-align' (Works for IE and Opera, but not Firefox). Also try <code>margin-right:auto; margin-left:0;</code> for left or <code>margin-right:0; margin-left:auto;</code> for right (optionally replacing 0 with the original margin for that side)</td></tr>
    <tr class="impl-yes"><td>H1, H2, H3, H4, H5, H6, P</td><td>Equivalent style 'text-align'</td></tr>
 <tr class="required impl-yes"><td>alt</td><td>IMG</td><td>Required, insert image filename if src is present or default invalid image text</td></tr>
-<tr><td rowspan="3">bgcolor</td><td>TABLE</td><td>Equivalent style 'background-color' (IE tested)</td></tr>
-    <tr><td>TR</td><td>Equivalent style 'background-color' (IE tested)</td></tr>
-    <tr><td>TD, TH</td><td>Equivalent style 'background-color'</td></tr>
-<tr><td>border</td><td>IMG</td><td>Equivalent style 'border-width', only applies when link present</td></tr>
+<tr class="impl-yes"><td rowspan="3">bgcolor</td><td>TABLE</td><td>Superset style 'background-color'</td></tr>
+    <tr class="impl-yes"><td>TR</td><td>Superset style 'background-color'</td></tr>
+    <tr class="impl-yes"><td>TD, TH</td><td>Superset style 'background-color'</td></tr>
+<tr class="impl-yes"><td>border</td><td>IMG</td><td>Equivalent style <code>border:[number]px solid</code></td></tr>
 <tr><td>clear</td><td>BR</td><td>Near-equiv style 'clear', transform 'all' into 'both'</td></tr>
 <tr class="impl-no"><td>compact</td><td>DL, OL, UL</td><td>Boolean, needs custom CSS class; rarely used anyway</td></tr>
 <tr class="required impl-yes"><td>dir</td><td>BDO</td><td>Required, insert ltr (or configuration value) if none</td></tr>
-<tr><td>height</td><td>TD, TH</td><td>Near-equiv style 'height', needs px suffix if original was in pixels</td></tr>
+<tr class="impl-yes"><td>height</td><td>TD, TH</td><td>Near-equiv style 'height', needs px suffix if original was in pixels</td></tr>
 <tr><td>hspace</td><td>IMG</td><td>Near-equiv styles 'margin-top' and 'margin-bottom', needs px suffix</td></tr>
 <tr class="impl-yes"><td>lang</td><td>*</td><td>Copy value to xml:lang</td></tr>
-<tr><td rowspan="2">name</td><td>IMG</td><td>Turn into ID</td></tr>
-    <tr><td>A</td><td>Turn into ID? (not deprecated, though in which specs?)</td></tr>
+<tr class="impl-yes"><td rowspan="2">name</td><td>IMG</td><td>Turn into ID</td></tr>
+    <tr class="impl-yes"><td>A</td><td>Turn into ID</td></tr>
 <tr><td>noshade</td><td>HR</td><td>Boolean, style 'border-style:solid;'</td></tr>
 <tr><td>nowrap</td><td>TD, TH</td><td>Boolean, style 'white-space:nowrap;' (not compat with IE5)</td></tr>
-<tr><td>size</td><td>HR</td><td>Near-equiv 'width', needs px suffix if original was pixels</td></tr>
+<tr><td>size</td><td>HR</td><td>Near-equiv 'height', needs px suffix if original was pixels</td></tr>
 <tr class="required impl-yes"><td>src</td><td>IMG</td><td>Required, insert blank or default img if not set</td></tr>
 <tr class="impl-yes"><td>start</td><td>OL</td><td>Poorly supported 'counter-reset', allowed in loose, dropped in strict</td></tr>
 <tr><td rowspan="3">type</td><td>LI</td><td rowspan="3">Equivalent style 'list-style-type', different allowed values though. (needs testing)</td></tr>
@@ -290,8 +291,8 @@ Mozilla on inside and needs -moz-outline, no IE support.</td></tr>
    <tr><td>UL</td></tr>
 <tr class="impl-yes"><td>value</td><td>LI</td><td>Poorly supported 'counter-reset', allowed in loose, dropped in strict</td></tr>
 <tr><td>vspace</td><td>IMG</td><td>Near-equiv styles 'margin-left' and 'margin-right', needs px suffix, see hspace</td></tr>
-<tr><td rowspan="2">width</td><td>HR</td><td rowspan="2">Near-equiv style 'width', needs px suffix if original was pixels</td></tr>
-    <tr><td>TD, TH</td></tr>
+<tr class="impl-yes"><td rowspan="2">width</td><td>HR</td><td rowspan="2">Near-equiv style 'width', needs px suffix if original was pixels</td></tr>
+    <tr class="impl-yes"><td>TD, TH</td></tr>
 </tbody>

 </table>
--- a/docs/enduser-id.html
+++ b/docs/enduser-id.html
@@ -15,6 +15,7 @@

 <div id="filing">Filed under End-User</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Prior to HTML Purifier 1.2.0, this library blithely accepted user input that
 looked like this:</p>
--- a/docs/enduser-overview.txt
+++ b/docs/enduser-overview.txt
@@ -36,7 +36,7 @@ forgiving lexer.  You may also be interested in the unit tests located in the
 tests/ folder, which provide a living document on how exactly the filter deals
 with malformed input.

-In summary:
+In summary (see corresponding classes for more details):

 1. Parse document into an array of tag and text tokens (Lexer)
 2. Remove all elements not on whitelist and transform certain other elements
--- a/docs/enduser-security.txt
+++ b/docs/enduser-security.txt
@@ -6,43 +6,17 @@ through negligence of people. This class will do its job: no more, no less,
 and it's up to you to provide it the proper information and proper context
 to be effective. Things to remember:

-1. Character Encoding: UTF-8.
-Currently, the parser runs under the assumption that it is dealing
-with UTF-8. Not ISO-8859-1 or Windows-1252, UTF-8. And definitely not "no
-character encoding explicitly stated" or UTF-7. If you're not using UTF-8 as
-your character encoding, make sure you configure HTML Purifier or switch
-to UTF-8. Now. Also, make sure any input is properly converted to UTF-8, or
-the parser will mangle it badly (though it won't be a security risk if you're
-outputting it as UTF-8 though).  Character encoding is, in general, a knotty
-issue, but do yourself a favor and learn about it:
-<http://www.joelonsoftware.com/articles/Unicode.html>
+1. Character Encoding: see enduser-utf8.html for more info.

-2. Doctype: XHTML 1.0 Transitional
-This is what the parser is outputting. For the most
-part, it's compatible with HTML 4.01, but XHTML enforces some very nice things
-that all web developers should use. Regardless, NO DOCTYPE is a NO. Quirks mode
-has waaaay too many quirks for a little parser to handle.  We did not select
-strict in order to prevent ourselves from being too draconic on users, but
-this may be configurable in the future.  Do you want standards compliance?
-The doctype is a good place to start.
+2. Doctype: document pending feature completion
+Not strictly necessary, actually. More in-depth discussion once we figure
+out how to get strict loose mode working.

-3. IDs
-They need to be unique, but without some knowledge of the
-rest of the document, it's difficult to know what's unique. %Attr.IDBlacklist
-needs to be set: we may want to consider disallowing IDs by default to
-save lazy programmers.
+3. IDs: see enduser-id.html for more info

-4. [PROJECTED] Links
-We're not going to try for spam protection (although
-some hooks for such a module might be nice) but we may offer the ability to
-only accept relative URLs. Pick the one that's right for you.
+4. Links: document pending feature completion
+Rudimentary blacklisting, we should also allow only relative URIs. We
+need a doc to explain the stuff.

-5. CSS
-While we can prevent the most flagrant cases from affecting your
-layout (such as absolutely positioned elements), no amount of code is going
-to protect your pages from being attacked by garish colors and plain old
-bad taste.  A neat feature would be the ability to define acceptable colors
-in a document, but that's not likely to be implemented for a while.  In the
-meantime, be sure to make sure that floated elements (permitted, since they
-can be quite useful) can't mess up your layout. Once again, we may want to
-disable this by default to protect lazy developers.
+5. CSS: document pending
+Explain which CSS styles we blocked and why.
--- a/docs/enduser-slow.html
+++ b/docs/enduser-slow.html
@@ -0,0 +1,117 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Strict//EN"
+    "http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd">
+<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en"><head>
+<meta http-equiv="Content-Type" content="text/html; charset=UTF-8" />
+<meta name="description" content="Explains how to speed up HTML Purifier through caching or inbound filtering." />
+<link rel="stylesheet" type="text/css" href="./style.css" />
+
+<title>Speeding up HTML Purifier - HTML Purifier</title>
+
+</head><body>
+
+<h1 class="subtitled">Speeding up HTML Purifier</h1>
+<div class="subtitle">...also known as the HELP ME LIBRARY IS TOO SLOW MY PAGE TAKE TOO LONG page</div>
+
+<div id="filing">Filed under End-User</div>
+<div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>
+
+<p>HTML Purifier is a very powerful library. But with power comes great 
+responsibility, in the form of longer execution times.  Remember, this 
+library isn't lightly grazing over submitted HTML: it's deconstructing 
+the whole thing, rigorously checking the parts, and then putting it back 
+together. </p>
+
+<p>So, if it so turns out that HTML Purifier is kinda too slow for outbound 
+filtering, you've got a few options: </p>
+
+<h2>Inbound filtering</h2>
+
+<p>Perform filtering of HTML when it's submitted by the user. Since the 
+user is already submitting something, an extra half a second tacked on 
+to the load time probably isn't going to be that huge of a problem.  
+Then, displaying the content is a simple a manner of outputting it 
+directly from your database/filesystem. The trouble with this method is 
+that your user loses the original text, and when doing edits, will be 
+handling the filtered text.  While this may be a good thing, especially 
+if you're using a WYSIWYG editor, it can also result in data-loss if a 
+user makes a typo. </p>
+
+<p>Example (non-functional):</p>
+
+<pre>&lt;?php
+    /**
+     * FORM SUBMISSION PAGE
+     * display_error($message) : displays nice error page with message
+     * display_success() : displays a nice success page
+     * display_form() : displays the HTML submission form
+     * database_insert($html) : inserts data into database as new row
+     */
+    if (!empty($_POST)) {
+        require_once '/path/to/library/HTMLPurifier.auto.php';
+        require_once 'HTMLPurifier.func.php';
+        $dirty_html = isset($_POST['html']) ? $_POST['html'] : false;
+        if (!$dirty_html) {
+            display_error('You must write some HTML!');
+        }
+        $html = HTMLPurifier($dirty_html);
+        database_insert($html);
+        display_success();
+        // notice that $dirty_html is *not* saved
+    } else {
+        display_form();
+    }
+?&gt;</pre>
+
+<h2>Caching the filtered output</h2>
+
+<p>Accept the submitted text and put it unaltered into the database, but 
+then also generate a filtered version and stash that in the database.  
+Serve the filtered version to readers, and the unaltered version to 
+editors.  If need be, you can invalidate the cache and have the cached 
+filtered version be regenerated on the first page view.  Pros? Full data 
+retention. Cons? It's more complicated, and opens other editors up to 
+XSS if they are using a WYSIWYG editor (to fix that, they'd have to be 
+able to get their hands on the *really* original text served in 
+plaintext mode). </p>
+
+<p>Example (non-functional):</p>
+
+<pre>&lt;?php
+    /**
+     * VIEW PAGE
+     * display_error($message) : displays nice error page with message
+     * cache_get($id) : retrieves HTML from fast cache (db or file)
+     * cache_insert($id, $html) : inserts good HTML into cache system
+     * database_get($id) : retrieves raw HTML from database
+     */
+    $id = isset($_GET['id']) ? (int) $_GET['id'] : false;
+    if (!$id) {
+        display_error('Must specify ID.');
+        exit;
+    }
+    $html = cache_get($id); // filesystem or database
+    if ($html === false) {
+        // cache didn't have the HTML, generate it
+        $raw_html = database_get($id);
+        require_once '/path/to/library/HTMLPurifier.auto.php';
+        require_once 'HTMLPurifier.func.php';
+        $html = HTMLPurifier($raw_html);
+        cache_insert($id, $html);
+    }
+    echo $html;
+?&gt;</pre>
+
+<h2>Summary</h2>
+
+<p>In short, inbound filtering is the simple option and caching is the
+robust option (albeit with bigger storage requirements). </p>
+
+<p>There is a third option, independent of the two we've discussed: profile 
+and optimize HTMLPurifier yourself. Be sure to report back your results 
+if you decide to do that! Especially if you port HTML Purifier to C++. 
+<tt>;-)</tt></p>
+
+</body>
+</html>
--- a/docs/enduser-utf8.html
+++ b/docs/enduser-utf8.html
--- a/docs/enduser-youtube.html
+++ b/docs/enduser-youtube.html
@@ -15,6 +15,7 @@

 <div id="filing">Filed under End-User</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Clients like their YouTube videos. It gives them a warm fuzzy feeling when
 they see a neat little embedded video player on their websites that can play
@@ -36,7 +37,7 @@ from a specific website, it probably is okay. If no amount of pleading will
 convince the people upstairs that they should just settle with just linking
 to their movies, you may find this technique very useful.</p>

-<h2>Sample</h2>
+<h2>Looking in</h2>

 <p>Below is custom code that allows users to embed
 YouTube videos. This is not favoritism: this trick can easily be adapted for
@@ -68,55 +69,27 @@ into your documents. YouTube's code goes like this:</p>
 <p>What point 2 means is that if we have code like <code>&lt;span
 class=&quot;embed-youtube&quot;&gt;AyPzM5WK8ys&lt;/span&gt;</code> your
 application can reconstruct the full object from this small snippet that
-passes through HTML Purifier <em>unharmed</em>.</p>
+passes through HTML Purifier <em>unharmed</em>.
+<a href="http://hp.jpsband.org/svnroot/htmlpurifier/trunk/library/HTMLPurifier/Filter/YouTube.php">Show me the code!</a></p>

-<pre>
-&lt;?php
+<p>And the corresponding usage:</p>

-class HTMLPurifierX_PreserveYouTube extends HTMLPurifier
-{
-    function purify($html, $config = null) {
-        $pre_regex = '#&lt;object[^&gt;]+&gt;.+?'.
-            'http://www.youtube.com/v/([A-Za-z0-9]+).+?&lt;/object&gt;#';
-        $pre_replace = '&lt;span class=&quot;youtube-embed&quot;&gt;\1&lt;/span&gt;';
-        $html = preg_replace($pre_regex, $pre_replace, $html);
-        $html = parent::purify($html, $config);
-        $post_regex = '#&lt;span class=&quot;youtube-embed&quot;&gt;([A-Za-z0-9]+)&lt;/span&gt;#';
-        $post_replace = '&lt;object width=&quot;425&quot; height=&quot;350&quot; '.
-            'data=&quot;http://www.youtube.com/v/\1&quot;&gt;'.
-            '&lt;param name=&quot;movie&quot; value=&quot;http://www.youtube.com/v/\1&quot;&gt;&lt;/param&gt;'.
-            '&lt;param name=&quot;wmode&quot; value=&quot;transparent&quot;&gt;&lt;/param&gt;'.
-            '&lt;!--[if IE]&gt;'.
-            '&lt;embed src=&quot;http://www.youtube.com/v/\1&quot;'.
-            'type=&quot;application/x-shockwave-flash&quot;'.
-            'wmode=&quot;transparent&quot; width=&quot;425&quot; height=&quot;350&quot; /&gt;'.
-            '&lt;![endif]--&gt;'.
-            '&lt;/object&gt;';
-        $html = preg_replace($post_regex, $post_replace, $html);
-        return $html;
-    }
-}
+<pre>&lt;?php
+    // assuming $purifier is an instance of HTMLPurifier
+    require_once 'HTMLPurifier/Filter/YouTube.php';
+    $purifier-&gt;addFilter(new HTMLPurifier_Filter_YouTube());
+?&gt;</pre>

-$purifier = new HTMLPurifierX_PreserveYouTube();
-$html_still_with_youtube = $purifier->purify($html_with_youtube);
-
-?&gt;
-</pre>
-
-<p>There is a bit going on here, so let's explain.</p>
+<p>There is a bit going in the two code snippets, so let's explain.</p>

 <ol>
-    <li>The class uses the prefix <code>HTMLPurifierX</code> because it's
-        userspace code. Don't use <code>HTMLPurifier</code> in front of your
-        class, since it might clobber another class in the library.</li>
-    <li>In order to keep the interface compatible, we've extended HTMLPurifier
-        into a new class that preserves the YouTube videos. This means that
-        all you have to do is replace all instances of
-        <code>new HTMLPurifier</code> to <code>new
-        HTMLPurifierX_PreserveYouTube</code>. There's other ways to go about
-        doing this: if you were calling a function that wrapped HTML Purifier,
-        you could paste the PHP right there. If you wanted to be really
-        fancy, you could make a decorator for HTMLPurifier.</li>
+    <li>This is a Filter object, which intercepts the HTML that is
+        coming into and out of the purifier. You can add as many
+        filter objects as you like. <code>preFilter()</code>
+        processes the code before it gets purified, and <code>postFilter()</code>
+        processes the code afterwards. So, we'll use <code>preFilter()</code> to
+        replace the object tag with a <code>span</code>, and <code>postFilter()</code>
+        to restore it.</li>
    <li>The first preg_replace call replaces any YouTube code users may have
        embedded into the benign span tag. Span is used because it is inline,
        and objects are inline too. We are very careful to be extremely
@@ -164,16 +137,16 @@ it is important that you are cognizant of the risk.</p>

 <p>This should go without saying, but if you're going to adapt this code
 for Google Video or the like, make sure you do it <em>right</em>. It's
-extremely easy to allow a character too many in the final section and
+extremely easy to allow a character too many in <code>postFilter()</code> and
 suddenly you're introducing XSS into HTML Purifier's XSS free output. HTML
 Purifier may be well written, but it cannot guard against vulnerabilities
 introduced after it has finished.</p>

-<h2>Future plans</h2>
+<h2>Help out!</h2>

-<p>It would probably be a good idea if this code was added to the core
-library. Look out for the inclusion of this into the core as a decorator
-or the like.</p>
+<p>If you write a filter for your favorite video destination (or anything
+like that, for that matter), send it over and it might get included
+with the core!</p>

 </body>
 </html>
--- a/docs/examples/basic.php
+++ b/docs/examples/basic.php
@@ -1,15 +1,14 @@
-<?php
+<?php exit;

 // This file demonstrates basic usage of HTMLPurifier.

-exit; // not to be called directly, it will fail fantastically!
-
-set_include_path('/path/to/htmlpurifier/library' . PATH_SEPARATOR . get_include_path());
-require_once 'HTMLPurifier.php';
+require_once '/path/to/htmlpurifier/library/HTMLPurifier.auto.php';

 $purifier = new HTMLPurifier();
 $html = '<b>Simple and short';

 $pure_html = $purifier->purify($html);

+echo $pure_html;
+
 ?>
--- a/docs/fixquotes.htc
+++ b/docs/fixquotes.htc
@@ -0,0 +1,6 @@
+<public:attach event="oncontentready" onevent="init();" />
+<script>
+function init() {
+  element.innerHTML = '&#8220;'+element.innerHTML+'&#8221;';
+}
+</script>
--- a/docs/index.html
+++ b/docs/index.html
@@ -13,7 +13,7 @@

 <h1>Documentation</h1>

-<p><strong>HTML Purifier</strong> has documentation for all types of people.
+<p><strong><a href="http://hp.jpsband.org/">HTML Purifier</a></strong> has documentation for all types of people.
 Here is an index of all of them.</p>

 <h2>End-user</h2>
@@ -28,6 +28,12 @@ information for casual developers using HTML Purifier.</p>
 <dt><a href="enduser-youtube.html">Embedding YouTube videos</a></dt>
 <dd>Explains how to safely allow the embedding of flash from trusted sites.</dd>

+<dt><a href="enduser-slow.html">Speeding up HTML Purifier</a></dt>
+<dd>Explains how to speed up HTML Purifier through caching or inbound filtering.</dd>
+
+<dt><a href="enduser-utf8.html">UTF-8: The Secret of Character Encoding</a></dt>
+<dd>Describes the rationale for using UTF-8, the ramifications otherwise, and how to make the switch.</dd>
+
 </dl>

 <h2>Development</h2>
@@ -48,6 +54,10 @@ conventions.</p>
 <dt><a href="dev-optimization.html">Optimization</a></dt>
 <dd>Discusses possible methods of optimizing HTML Purifier.</dd>

+<dt><a href="dev-advanced-api.html">Advanced API</a></dt>
+<dd>Functional specification for HTML Purifier's advanced API for defining
+custom filtering behavior.</dd>
+
 </dl>

 <h2>Proposals</h2>
--- a/docs/proposal-colors.html
+++ b/docs/proposal-colors.html
@@ -15,6 +15,7 @@

 <div id="filing">Filed under Proposals</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Your website probably has a color-scheme.
 <span style="color:#090; background:#FFF;">Green on white</span>,
--- a/docs/proposal-config.txt
+++ b/docs/proposal-config.txt
@@ -7,22 +7,22 @@ value is used for.  This means decentralized configuration declarations that
 are nevertheless error checking and a centralized configuration object.

 Directives are divided into namespaces, indicating the major portion of
-functionality they cover (although there may be overlaps.  Please consult
+functionality they cover (although there may be overlaps).  Please consult
 the documentation in ConfigDef for more information on these namespaces.

 Since configuration is dependant on context, internal classes require a
 configuration object to be passed as a parameter.  (They also require a
 Context object).

-In relation to HTMLDefinition and CSSDefinition, there is a special class
+In relation to HTMLDefinition and CSSDefinition, there could be a special class
 of directives that influence the *construction* of the Definition object.
-A standard call pattern would look like:
+A theoretical call pattern would look like:

 1. Client calls Config->getHTMLDefinition()
 2. Config calls HTMLDefinition->createNew(this)
 3. HTMLDefinition constructs itself with base configuration
-4. HTMLDefinition calls Config->get('HTMLDefinition')
-5. Config returns array of directives that later construction
+4. HTMLDefinition calls Config->get('HTML')
+5. Config returns array of directives
 6. HTMLDefinition performs operations and changes specified by directives
 7. HTMLPurifier returns constructed definition
 8. Config caches definition so it doesn't have to be generated again
@@ -33,3 +33,8 @@ custom copy, which OVERRIDES all directives.  Only the base, vanilla copy
 is the Singleton, the object actually interfaced with is a operated-upon
 clone of that object.  Also, if an update to the directives would update
 the definition, you'd have to force reconstruction.
+
+In practice, the pulling directives from the config object are
+solely need-based, and the flex points are littered throughout the
+setup() function.  Some sort of refactoring is likely in order. See
+ref-xhtml-1.1.txt for more info.
--- a/docs/proposal-filter-levels.txt
+++ b/docs/proposal-filter-levels.txt
@@ -15,7 +15,10 @@ and properties to allow. HTMLDefinition makes a big part of what HTMLPurifier
 is. 

 The idea, then, is to setup fundamentally different set of definitions, which
-can further be customized using simpler configuration options.
+can further be customized using simpler configuration options.  Alternatively,
+they could be implemented as configuration profiles, which simply load
+a set of recommended directives to acheive a desired affect (no simpler
+config options though).

 Here are some fuzzy levels you could set:

--- a/docs/proposal-language.txt
+++ b/docs/proposal-language.txt
@@ -1,42 +1,6 @@
 We are going to model our I18N/L10N off of MediaWiki's system.  Their's is
 obviously quite complicated, so we're going to simplify it a bit for our needs.

-== Structure ==
-
-First, you have a Language object.  This object contains all the localisable
-message strings, as well as other important language-specific settings and
-custom behavior (uppercasing, lowercasing, printing dates, formatting
-numbers, etc.)
-
-The object is constructed from two sources: subclassed versions of itself
-(classes) and Message files (messages).
-
-== General use ==
-
-You load a language object by calling the Language::factory() function. 
-This function the class file for the object (taking in account fallback 
-languages by using the fallback langauge's object but overloading the 
-language key) and returns that object. Nothing else happens.
-
-When a message/etc is requested, a lazy load initializor is called.  Now the
-real work starts.  We're first going to take the scenario that the language
-is not cached.  The system loads the Messages file by:
-
-    require( $filename );
-    $cache = compact( self::$mLocalisationKeys );	
-
-...where self::$mLocalisationKeys is the name of variables that could be used
-in the localization file. This lets you use things like:
-
-    $fallback = false;
-    $rtl = false;
-
-...and easily siphon them into arrays.
-
-Then, we load the $fallback language (if not set, English) to fill in the gaps in
-the messages.  There is specialized behavior for certain keys, as they can be
-mergeable maps, lists or alias lists (not sure what the last one is).
-
 == Caching ==

 MediaWiki has lots of caching mechanisms built in, which make the code somewhat
--- a/docs/proposal-new-directives.txt
+++ b/docs/proposal-new-directives.txt
@@ -4,8 +4,6 @@ Configuration Ideas
 Here are some theoretical configuration ideas that we could implement some
 time.  Note the naming convention: %Namespace.Directive

-%Attr.IDPrefix - prefix all ids with this
-
 %Attr.RewriteFragments - if there's %Attr.IDPrefix we may want to transparently
    rewrite the URLs we parse too.  However, we can only do it when it's a pure
    anchor link, so it's not foolproof
--- a/docs/ref-devnetwork.html
+++ b/docs/ref-devnetwork.html
@@ -15,6 +15,7 @@

 <div id="filing">Filed under Reference</div>
 <div id="index">Return to the <a href="index.html">index</a>.</div>
+<div id="home"><a href="http://hp.jpsband.org/">HTML Purifier</a> End-User Documentation</div>

 <p>Many thanks to the DevNetwork community for answering questions,
 theorizing about design, and offering encouragement during
--- a/docs/ref-loose-vs-strict.txt
+++ b/docs/ref-loose-vs-strict.txt
@@ -32,6 +32,6 @@ A tag's attribute 'target' (for selecting frames) cut
    current behavior: no substitute, just delete when in strict, allow in loose
 Attribute 'name' deprecated in favor of 'id'
    current behavior: dropped silently
-    projected behavior: create proper AttrTransform (currently not allowed at all)
+    projected behavior: create proper AttrTransform
 [done] PRE tag allows SUB/SUP? (strict dtd comment vs syntax, loose disallows)
    current behavior: disallow as usual
--- a/docs/ref-strictness.txt
+++ b/docs/ref-strictness.txt
@@ -2,8 +2,8 @@
 Is HTML Purifier Strict or Transitional?
    A little bit of helpful guidance

-Despite the fact that HTML Purifier professes only to support transitional
-HTML, it rejects a lot of attributes and elements that are actually, indeed,
+Despite the fact that HTML Purifier professes to support both transitional and
+strict HTML, it rejects a lot of attributes and elements that are actually, indeed,
 valid. You can investigate progress.html to find out precisely what we
 are doing to these *deprecated* attributes.

@@ -11,8 +11,8 @@ However, users have found that Strict HTML imposes some quite unreasonable
 restrictions on certain things. The start and value attributes in ol and
 li (respectively) perhaps are the most contested. There's is currently no
 widely supported browser method short of JavaScript that can replace these
-two deprecated elements. HTML Purifier does not currently support them, but
-it might behoove us to do so while our output is still transitional.
+two deprecated elements. It behooves us to allow these deprecated
+attributes when the output is transitional.

 Fortunantely, that's the only real bugger case. The others have near-perfect
 CSS equivalents, and were presentational anyway. However, the other question
@@ -32,5 +32,6 @@ these loose-only constructs in loose mode:

 The changed child definitions as well as the ul.start li.value are the most
 compelling reasons why loose should be used.  We may want offer disabling <u>,
-<strike> and <s> by themselves.
+<strike> and <s> by themselves. We may also want to offer no pre-emptive
+deprecated conversions. This all must be unified.

--- a/docs/ref-xhtml-1.1.txt
+++ b/docs/ref-xhtml-1.1.txt
@@ -1,21 +1,187 @@

-Getting XHTML 1.1 Working
-
-It's quite simple, according to <http://www.w3.org/TR/xhtml11/changes.html>
+XHTML 1.1 and HTML Purifier

+Todo for XHTML 1.1 support <http://www.w3.org/TR/xhtml11/changes.html>
 1. Scratch lang entirely in favor of xml:lang
 2. Scratch name entirely in favor of id (partially-done)
 3. Support Ruby <http://www.w3.org/TR/2001/REC-ruby-20010531/>

-...but that's only an informative section. More things to do:
+HTML Purifier uses the modularization of XHTML
+<http://www.w3.org/TR/xhtml-modularization/> to organize the internals
+of HTMLDefinition into a more manageable and extensible fashion. Rather
+than have one super-object, HTMLDefinition is split into HTMLModules,
+each of which are responsible for defining elements, their attributes,
+and other properties (for a more indepth coverage, see
+/library/HTMLPurifier/HTMLModule.php's docblock comments).

-1. Scratch style attribute (it's deprecated)
-2. Be module-aware (this might entail intelligent grouping in the definition
-   and allowing users to specifically remove certain modules (see 5))
-3. Cross-reference minimal content models with existing DTDs and determine
-   changes (todo)
-4. Watch out for the Legacy Module
-<http://www.w3.org/TR/2001/REC-xhtml-modularization-20010410/abstract_modules.html#s_legacymodule>
-5. Let users specify their own custom modules
-6. Study Modularization document
-<http://www.w3.org/TR/2001/REC-xhtml-modularization-20010410/>
+The modules that W3C defines and we support are:
+
+    * 5.1. Attribute Collections (technically not a module
+    * 5.2. Core Modules
+          o 5.2.2. Text Module
+          o 5.2.3. Hypertext Module
+          o 5.2.4. List Module
+    * 5.4. Text Extension Modules
+          o 5.4.1. Presentation Module
+          o 5.4.2. Edit Module
+          o 5.4.3. Bi-directional Text Module
+    * 5.6. Table Modules
+          o 5.6.2. Tables Module
+    * 5.7. Image Module
+    * 5.18. Style Attribute Module
+
+Modules that we don't support but coul support are:
+
+    * 5.6. Table Modules
+          o 5.6.1. Basic Tables Module [?]
+    * 5.8. Client-side Image Map Module [?]
+    * 5.9. Server-side Image Map Module [?]
+    * 5.12. Target Module [?]
+    * 5.21. Name Identification Module [deprecated]
+    * 5.22. Legacy Module [deprecated]
+
+These modules will not be implemented due to their dangerousness or
+inapplicability as an XHTML fragment:
+
+    * 5.2. Core Modules
+          o 5.2.1. Structure Module
+    * 5.3. Applet Module
+    * 5.5. Forms Modules
+          o 5.5.1. Basic Forms Module
+          o 5.5.2. Forms Module
+    * 5.10. Object Module
+    * 5.11. Frames Module
+    * 5.13. Iframe Module
+    * 5.14. Intrinsic Events Module
+    * 5.15. Metainformation Module
+    * 5.16. Scripting Module
+    * 5.17. Style Sheet Module
+    * 5.19. Link Module
+    * 5.20. Base Module
+
+We will not be using W3C's XML Schemas or DTDs directly due to the lack
+of robust tools for handling them (the main problem is that all the
+current parsers are usually PHP 5 only and solely-validating, not
+correcting).
+
+The abstraction of the HTMLDefinition creation process will also
+contribute to a need for a caching system. Cache invalidation would be
+difficult, but could be done by comparing the HTML and Attr config
+namespaces with a copy that was packaged along with the serialized
+HTMLDefinition object.
+
+== General Use-Case ==
+
+The outwards API of HTMLDefinition has been largely preserved, not
+only for backwards-compatibility but also by design. Instead,
+HTMLDefinition can be retrieved "raw", in which it loads a structure
+that closely resembles the modules of XHTML 1.1. This structure is very
+dynamic, making it easy to make cascading changes to global content
+sets or remove elements in bulk.
+
+However, once HTML Purifier needs the actual definition, it retrieves
+a finalized version of HTMLDefinition. The finalized definition involves
+processing the modules into a form that it is optimized for multiple
+calls. This final version is immutable and, even if editable, would
+be extremely hard to change.
+
+So, some code taking advantage of the XHTML modularization may look
+like this:
+
+<?php
+    $config = HTMLPurifier_Config::createDefault();
+    $def =& $config->getHTMLDefinition(true); // reference to raw
+    unset($def->modules['Hypertext']); // rm ''a'' link
+    $purifier = new HTMLPurifier($config);
+    $purifier->purify($html); // now the definition is finalized
+?>
+
+== Inclusions ==
+
+One of the nice features of HTMLDefinition is that piggy-backing off
+of global attribute and content sets is extremely easy to do.
+
+=== Attributes ===
+
+HTMLModule->elements[$element]->attr stores attribute information for the
+specific attributes of $element. This is quite close to the final
+API that HTML Purifier interfaces with, but there's an important
+extra feature: attr may also contain a array with a member index zero.
+
+<?php
+    HTMLModule->elements[$element]->attr[0] = array('AttrSet');
+?>
+
+Rather than map the attribute key 0 to an array (which should be
+an AttrDef), it defines a number of attribute collections that should
+be merged into this elements attribute array.
+
+Furthermore, the value of an attribute key, attribute value pair need
+not be a fully fledged AttrDef object. They can also be a string, which
+signifies a AttrDef that is looked up from a centralized registry
+AttrTypes. This allows more concise attribute definitions that look
+more like W3C's declarations, as well as offering a centralized point
+for modifying the behavior of one attribute type. And, of course, the
+old method of manually instantiating an AttrDef still works.
+
+=== Attribute Collections ===
+
+Attribute collections are stored and processed in the AttrCollections
+object, which is responsible for performing the inclusions signified
+by the 0 index. These attribute collections, too, are mutable, by
+using HTMLModule->attr_collections. You may add new attributes
+to a collection or define an entirely new collection for your module's
+use. Inclusions can also be cumulative.
+
+Attribute collections allow us to get rid of so called "global attributes"
+(which actually aren't so global).
+
+=== Content Models and ChildDef ===
+
+An implementation of the above-mentioned attributes and attribute
+collections was applied to the ChildDef system. HTML Purifier uses
+a proprietary system called ChildDef for performance and flexibility
+reasons, but this does not line up very well with W3C's notion of
+regexps for defining the allowed children of an element.
+
+HTMLPurifier->elements[$element]->content_model and 
+HTMLPurifier->elements[$element]->content_model_type store information
+about the final ChildDef that will be stored in
+HTMLPurifier->elements[$element]->child (we use a different variable
+because the two forms are sufficiently different).
+
+$content_model is an abstract, string representation of the internal
+state of ChildDef, while $content_model_type is a string identifier
+of which ChildDef subclass to instantiate. $content_model is processed
+by substituting all content set identifiers (capitalized element names)
+with their contents. It is then parsed and passed into the appropriate
+ChildDef class, as defined by the ContentSets->getChildDef() or the
+custom fallback HTMLModule->getChildDef() for custom child definitions
+not in the core.
+
+You'll need to use these facilities if you plan on referencing a content
+set like "Inline" or "Block", and using them is recommended even if you're
+not due to their conciseness.
+
+A few notes on $content_model: it's structure can be as complicated
+as you want, but the pipe symbol (|) is reserved for defining possible
+choices, due to the content sets implementation. For example, a content
+model that looks like:
+
+"Inline -> Block -> a"
+
+...when the Inline content set is defined as "span | b" and the Block
+content set is defined as "div | blockquote", will expand into:
+
+"span | b -> div | blockquote -> a"
+
+The custom HTMLModule->getChildDef() function will need to be able to
+then feed this information to ChildDef in a usable manner.
+
+=== Content Sets ===
+
+Content sets can be altered using HTMLModule->content_sets, an associative
+array of content set names to content set contents. If the content set
+already exists, your values are appended on to it (great for, say,
+registering the font tag as an inline element), otherwise it is
+created. They are substituted into content_model.
--- a/docs/style.css
+++ b/docs/style.css
@@ -23,6 +23,8 @@ h4 {font-family:sans-serif; font-size:0.9em; font-weight:bold; }

 /* Marks off asides, discussions on why something is the way it is */
 .aside {margin-left:2em; font-family:sans-serif; font-size:0.9em; }
+blockquote .label {font-weight:bold; font-size:1em; margin:0 0 .1em;
+    border-bottom:1px solid #CCC;}

 /* A regular table */
 .table {border-collapse:collapse; border-bottom:2px solid #888; margin-left:2em; }
@@ -36,5 +38,31 @@ h4 {font-family:sans-serif; font-size:0.9em; font-weight:bold; }
 /* Contains, without exception, Return to index. */
 #index {font-size:smaller; }

+#home {font-size:smaller;}
+
 /* Contains, without exception, $Id$, for SVN version info. */
-#version {text-align:right; font-style:italic; margin:2em 0;}
+#version {text-align:right; font-style:italic; margin:2em 0;}
+
+#toc ol ol {list-style-type:lower-roman;}
+#toc ol {list-style-type:decimal;}
+#toc {list-style-type:upper-alpha;}
+
+q {
+  behavior: url(fixquotes.htc); /* IE fix */
+  quotes: '\201C' '\201D' '\2018' '\2019';
+}
+q:before {
+  content: open-quote;
+}
+q:after {
+  content: close-quote;
+}
+
+/* Marks off implementation details interesting only to the person writing
+   the class described in the spec. */
+.technical {margin-left:2em; }
+.technical:before {content:"Technical note: "; font-weight:bold; color:#061; }
+
+/* Marks off sections that are lacking. */
+.fixme {margin-left:2em; }
+.fixme:before {content:"Fix me: "; font-weight:bold; color:#C00; }
--- a/library/HTMLPurifier.func.php
+++ b/library/HTMLPurifier.func.php
@@ -6,12 +6,12 @@
 *       this is efficient for instances when you only use HTML Purifier
 *       on a few of your pages, it murders bytecode caching. You still
 *       need to add HTML Purifier to your path.
+ * @note ''HTMLPurifier()'' is NOT the same as ''new HTMLPurifier()''
 */

 function HTMLPurifier($html, $config = null) {
    static $purifier = false;
    if (!$purifier) {
-        $init = true;
        require_once 'HTMLPurifier.php';
        $purifier = new HTMLPurifier();
    }
--- a/library/HTMLPurifier.php
+++ b/library/HTMLPurifier.php
@@ -22,7 +22,7 @@
 */

 /*
-    HTML Purifier 1.3.2 - Standards Compliant HTML Filtering
+    HTML Purifier 1.6.0 - Standards Compliant HTML Filtering
    Copyright (C) 2006 Edward Z. Yang

    This library is free software; you can redistribute it and/or
@@ -64,9 +64,10 @@ require_once 'HTMLPurifier/Encoder.php';
 class HTMLPurifier
 {
    
-    var $version = '1.3.2';
+    var $version = '1.6.0';
    
    var $config;
+    var $filters;
    
    var $lexer, $strategy, $generator;
    
@@ -91,10 +92,17 @@ class HTMLPurifier
        $this->lexer        = HTMLPurifier_Lexer::create();
        $this->strategy     = new HTMLPurifier_Strategy_Core();
        $this->generator    = new HTMLPurifier_Generator();
-        $this->encoder      = new HTMLPurifier_Encoder();
        
    }
    
+    /**
+     * Adds a filter to process the output. First come first serve
+     * @param $filter HTMLPurifier_Filter object
+     */
+    function addFilter($filter) {
+        $this->filters[] = $filter;
+    }
+    
    /**
     * Filters an HTML snippet/document to be XSS-free and standards-compliant.
     * 
@@ -109,8 +117,12 @@ class HTMLPurifier
        
        $config = $config ? HTMLPurifier_Config::create($config) : $this->config;
        
-        $context =& new HTMLPurifier_Context();
-        $html = $this->encoder->convertToUTF8($html, $config, $context);
+        $context = new HTMLPurifier_Context();
+        $html = HTMLPurifier_Encoder::convertToUTF8($html, $config, $context);
+        
+        for ($i = 0, $size = count($this->filters); $i < $size; $i++) {
+            $html = $this->filters[$i]->preFilter($html, $config, $context);
+        }
        
        // purified HTML
        $html = 
@@ -127,7 +139,11 @@ class HTMLPurifier
                $config, $context
            );
        
-        $html = $this->encoder->convertFromUTF8($html, $config, $context);
+        for ($i = $size - 1; $i >= 0; $i--) {
+            $html = $this->filters[$i]->postFilter($html, $config, $context);
+        }
+        
+        $html = HTMLPurifier_Encoder::convertFromUTF8($html, $config, $context);
        $this->context =& $context;
        return $html;
    }
--- a/library/HTMLPurifier/AttrCollections.php
+++ b/library/HTMLPurifier/AttrCollections.php
@@ -0,0 +1,100 @@
+<?php
+
+require_once 'HTMLPurifier/AttrTypes.php';
+require_once 'HTMLPurifier/AttrDef/Lang.php';
+
+/**
+ * Defines common attribute collections that modules reference
+ */
+
+class HTMLPurifier_AttrCollections
+{
+    
+    /**
+     * Associative array of attribute collections, indexed by name
+     * @note Technically, the composition of these is more complicated,
+     *       but we bypass it using our own excludes property
+     */
+    var $info = array();
+    
+    /**
+     * Performs all expansions on internal data for use by other inclusions
+     * It also collects all attribute collection extensions from
+     * modules
+     * @param $attr_types HTMLPurifier_AttrTypes instance
+     * @param $modules Hash array of HTMLPurifier_HTMLModule members
+     */
+    function HTMLPurifier_AttrCollections($attr_types, $modules) {
+        $info =& $this->info;
+        // load extensions from the modules
+        foreach ($modules as $module) {
+            foreach ($module->attr_collections as $coll_i => $coll) {
+                foreach ($coll as $attr_i => $attr) {
+                    if ($attr_i === 0 && isset($info[$coll_i][$attr_i])) {
+                        // merge in includes
+                        $info[$coll_i][$attr_i] = array_merge(
+                            $info[$coll_i][$attr_i], $attr);
+                        continue;
+                    }
+                    $info[$coll_i][$attr_i] = $attr;
+                }
+            }
+        }
+        // perform internal expansions and inclusions
+        foreach ($info as $name => $attr) {
+            // merge attribute collections that include others
+            $this->performInclusions($info[$name]);
+            // replace string identifiers with actual attribute objects
+            $this->expandIdentifiers($info[$name], $attr_types);
+        }
+    }
+    
+    /**
+     * Takes a reference to an attribute associative array and performs
+     * all inclusions specified by the zero index.
+     * @param &$attr Reference to attribute array
+     */
+    function performInclusions(&$attr) {
+        if (!isset($attr[0])) return;
+        $merge = $attr[0];
+        // loop through all the inclusions
+        for ($i = 0; isset($merge[$i]); $i++) {
+            // foreach attribute of the inclusion, copy it over
+            foreach ($this->info[$merge[$i]] as $key => $value) {
+                if (isset($attr[$key])) continue; // also catches more inclusions
+                $attr[$key] = $value;
+            }
+            if (isset($info[$merge[$i]][0])) {
+                // recursion
+                $merge = array_merge($merge, isset($info[$merge[$i]][0]));
+            }
+        }
+        unset($attr[0]);
+    }
+    
+    /**
+     * Expands all string identifiers in an attribute array by replacing
+     * them with the appropriate values inside HTMLPurifier_AttrTypes
+     * @param &$attr Reference to attribute array
+     * @param $attr_types HTMLPurifier_AttrTypes instance
+     */
+    function expandIdentifiers(&$attr, $attr_types) {
+        foreach ($attr as $def_i => $def) {
+            if ($def_i === 0) continue;
+            if (!is_string($def)) continue;
+            if ($def === false) {
+                unset($attr[$def_i]);
+                continue;
+            }
+            if (isset($attr_types->info[$def])) {
+                $attr[$def_i] = $attr_types->info[$def];
+            } else {
+                trigger_error('Attempted to reference undefined attribute type', E_USER_ERROR);
+                unset($attr[$def_i]);
+            }
+        }
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/CSS.php
+++ b/library/HTMLPurifier/AttrDef/CSS.php
@@ -8,6 +8,11 @@ require_once 'HTMLPurifier/CSSDefinition.php';
 * @note We don't implement the whole CSS specification, so it might be
 *       difficult to reuse this component in the context of validating
 *       actual stylesheet declarations.
+ * @note If we were really serious about validating the CSS, we would
+ *       tokenize the styles and then parse the tokens. Obviously, we
+ *       are not doing that. Doing that could seriously harm performance,
+ *       but would make these components a lot more viable for a CSS
+ *       filtering solution.
 */
 class HTMLPurifier_AttrDef_CSS extends HTMLPurifier_AttrDef
 {
@@ -20,6 +25,9 @@ class HTMLPurifier_AttrDef_CSS extends HTMLPurifier_AttrDef
        
        // we're going to break the spec and explode by semicolons.
        // This is because semicolon rarely appears in escaped form
+        // Doing this is generally flaky but fast
+        // IT MIGHT APPEAR IN URIs, see HTMLPurifier_AttrDef_CSSURI
+        // for details
        
        $declarations = explode(';', $css);
        $propvalues = array();
--- a/library/HTMLPurifier/AttrDef/CSS/Background.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Background.php
@@ -0,0 +1,87 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef.php';
+require_once 'HTMLPurifier/CSSDefinition.php';
+
+/**
+ * Validates shorthand CSS property background.
+ * @warning Does not support url tokens that have internal spaces.
+ */
+class HTMLPurifier_AttrDef_CSS_Background extends HTMLPurifier_AttrDef
+{
+    
+    /**
+     * Local copy of component validators.
+     * @note See HTMLPurifier_AttrDef_Font::$info for a similar impl.
+     */
+    var $info;
+    
+    function HTMLPurifier_AttrDef_CSS_Background($config) {
+        $def = $config->getCSSDefinition();
+        $this->info['background-color'] = $def->info['background-color'];
+        $this->info['background-image'] = $def->info['background-image'];
+        $this->info['background-repeat'] = $def->info['background-repeat'];
+        $this->info['background-attachment'] = $def->info['background-attachment'];
+        $this->info['background-position'] = $def->info['background-position'];
+    }
+    
+    function validate($string, $config, &$context) {
+        
+        // regular pre-processing
+        $string = $this->parseCDATA($string);
+        if ($string === '') return false;
+        
+        // assumes URI doesn't have spaces in it
+        $bits = explode(' ', strtolower($string)); // bits to process
+        
+        $caught = array();
+        $caught['color']    = false;
+        $caught['image']    = false;
+        $caught['repeat']   = false;
+        $caught['attachment'] = false;
+        $caught['position'] = false;
+        
+        $i = 0; // number of catches
+        $none = false;
+        
+        foreach ($bits as $bit) {
+            if ($bit === '') continue;
+            foreach ($caught as $key => $status) {
+                if ($key != 'position') {
+                    if ($status !== false) continue;
+                    $r = $this->info['background-' . $key]->validate($bit, $config, $context);
+                } else {
+                    $r = $bit;
+                }
+                if ($r === false) continue;
+                if ($key == 'position') {
+                    if ($caught[$key] === false) $caught[$key] = '';
+                    $caught[$key] .= $r . ' ';
+                } else {
+                    $caught[$key] = $r;
+                }
+                $i++;
+                break;
+            }
+        }
+        
+        if (!$i) return false;
+        if ($caught['position'] !== false) {
+            $caught['position'] = $this->info['background-position']->
+                validate($caught['position'], $config, $context);
+        }
+        
+        $ret = array();
+        foreach ($caught as $value) {
+            if ($value === false) continue;
+            $ret[] = $value;
+        }
+        
+        if (empty($ret)) return false;
+        return implode(' ', $ret);
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/CSS/BackgroundPosition.php
+++ b/library/HTMLPurifier/AttrDef/CSS/BackgroundPosition.php
@@ -0,0 +1,130 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Length.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Percentage.php';
+
+/* W3C says:
+    [ // adjective and number must be in correct order, even if
+      // you could switch them without introducing ambiguity.
+      // some browsers support that syntax
+        [
+            <percentage> | <length> | left | center | right
+        ]
+        [ 
+            <percentage> | <length> | top | center | bottom
+        ]?
+    ] |
+    [ // this signifies that the vertical and horizontal adjectives
+      // can be arbitrarily ordered, however, there can only be two,
+      // one of each, or none at all
+        [
+            left | center | right
+        ] ||
+        [
+            top | center | bottom
+        ]
+    ]
+    top, left = 0%
+    center, (none) = 50%
+    bottom, right = 100%
+*/
+
+/* QuirksMode says:
+    keyword + length/percentage must be ordered correctly, as per W3C
+    
+    Internet Explorer and Opera, however, support arbitrary ordering. We
+    should fix it up.
+    
+    Minor issue though, not strictly necessary.
+*/
+
+// control freaks may appreciate the ability to convert these to
+// percentages or something, but it's not necessary
+
+/**
+ * Validates the value of background-position.
+ */
+class HTMLPurifier_AttrDef_CSS_BackgroundPosition extends HTMLPurifier_AttrDef
+{
+    
+    var $length;
+    var $percentage;
+    
+    function HTMLPurifier_AttrDef_CSS_BackgroundPosition() {
+        $this->length     = new HTMLPurifier_AttrDef_CSS_Length();
+        $this->percentage = new HTMLPurifier_AttrDef_CSS_Percentage();
+    }
+    
+    function validate($string, $config, &$context) {
+        $string = $this->parseCDATA($string);
+        $bits = explode(' ', $string);
+        
+        $keywords = array();
+        $keywords['h'] = false; // left, right
+        $keywords['v'] = false; // top, bottom
+        $keywords['c'] = false; // center
+        $measures = array();
+        
+        $i = 0;
+        
+        $lookup = array(
+            'top' => 'v',
+            'bottom' => 'v',
+            'left' => 'h',
+            'right' => 'h',
+            'center' => 'c'
+        );
+        
+        foreach ($bits as $bit) {
+            if ($bit === '') continue;
+            
+            // test for keyword
+            $lbit = ctype_lower($bit) ? $bit : strtolower($bit);
+            if (isset($lookup[$lbit])) {
+                $status = $lookup[$lbit];
+                $keywords[$status] = $lbit;
+                $i++;
+            }
+            
+            // test for length
+            $r = $this->length->validate($bit, $config, $context);
+            if ($r !== false) {
+                $measures[] = $r;
+                $i++;
+            }
+            
+            // test for percentage
+            $r = $this->percentage->validate($bit, $config, $context);
+            if ($r !== false) {
+                $measures[] = $r;
+                $i++;
+            }
+            
+        }
+        
+        if (!$i) return false; // no valid values were caught
+        
+        
+        $ret = array();
+        
+        // first keyword
+        if     ($keywords['h'])     $ret[] = $keywords['h'];
+        elseif (count($measures))   $ret[] = array_shift($measures);
+        elseif ($keywords['c']) {
+            $ret[] = $keywords['c'];
+            $keywords['c'] = false; // prevent re-use: center = center center
+        }
+        
+        if     ($keywords['v'])     $ret[] = $keywords['v'];
+        elseif (count($measures))   $ret[] = array_shift($measures);
+        elseif ($keywords['c'])     $ret[] = $keywords['c'];
+        
+        if (empty($ret)) return false;
+        return implode(' ', $ret);
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/CSS/Border.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Border.php
@@ -5,7 +5,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 /**
 * Validates the border property as defined by CSS.
 */
-class HTMLPurifier_AttrDef_Border extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Border extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -13,7 +13,7 @@ class HTMLPurifier_AttrDef_Border extends HTMLPurifier_AttrDef
     */
    var $info = array();
    
-    function HTMLPurifier_AttrDef_Border($config) {
+    function HTMLPurifier_AttrDef_CSS_Border($config) {
        $def = $config->getCSSDefinition();
        $this->info['border-width'] = $def->info['border-width'];
        $this->info['border-style'] = $def->info['border-style'];
--- a/library/HTMLPurifier/AttrDef/CSS/Color.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Color.php
@@ -5,7 +5,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 /**
 * Validates Color as defined by CSS.
 */
-class HTMLPurifier_AttrDef_Color extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Color extends HTMLPurifier_AttrDef
 {
    
    /**
--- a/library/HTMLPurifier/AttrDef/CSS/Composite.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Composite.php
@@ -9,7 +9,7 @@
 * especially useful for CSS values, which often are a choice between
 * an enumerated set of predefined values or a flexible data type.
 */
-class HTMLPurifier_AttrDef_Composite extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Composite extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -21,7 +21,7 @@ class HTMLPurifier_AttrDef_Composite extends HTMLPurifier_AttrDef
    /**
     * @param $defs List of HTMLPurifier_AttrDef objects
     */
-    function HTMLPurifier_AttrDef_Composite($defs) {
+    function HTMLPurifier_AttrDef_CSS_Composite($defs) {
        $this->defs = $defs;
    }
    
--- a/library/HTMLPurifier/AttrDef/CSS/Font.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Font.php
@@ -5,7 +5,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 /**
 * Validates shorthand CSS property font.
 */
-class HTMLPurifier_AttrDef_Font extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Font extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -30,7 +30,7 @@ class HTMLPurifier_AttrDef_Font extends HTMLPurifier_AttrDef
        'status-bar' => true
    );
    
-    function HTMLPurifier_AttrDef_Font($config) {
+    function HTMLPurifier_AttrDef_CSS_Font($config) {
        $def = $config->getCSSDefinition();
        $this->info['font-style']   = $def->info['font-style'];
        $this->info['font-variant'] = $def->info['font-variant'];
--- a/library/HTMLPurifier/AttrDef/CSS/FontFamily.php
+++ b/library/HTMLPurifier/AttrDef/CSS/FontFamily.php
@@ -7,7 +7,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 /**
 * Validates a font family list according to CSS spec
 */
-class HTMLPurifier_AttrDef_FontFamily extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_FontFamily extends HTMLPurifier_AttrDef
 {
    
    /**
--- a/library/HTMLPurifier/AttrDef/CSS/Length.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Length.php
@@ -1,13 +1,12 @@
 <?php

 require_once 'HTMLPurifier/AttrDef.php';
-require_once 'HTMLPurifier/AttrDef/Number.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Number.php';

 /**
 * Represents a Length as defined by CSS.
- * @warning Be sure not to confuse this with HTMLPurifier_AttrDef_Length!
 */
-class HTMLPurifier_AttrDef_CSSLength extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Length extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -26,8 +25,8 @@ class HTMLPurifier_AttrDef_CSSLength extends HTMLPurifier_AttrDef
     * @param $non_negative Bool indication whether or not negative values are
     *                      allowed.
     */
-    function HTMLPurifier_AttrDef_CSSLength($non_negative = false) {
-        $this->number_def = new HTMLPurifier_AttrDef_Number($non_negative);
+    function HTMLPurifier_AttrDef_CSS_Length($non_negative = false) {
+        $this->number_def = new HTMLPurifier_AttrDef_CSS_Number($non_negative);
    }
    
    function validate($length, $config, &$context) {
@@ -40,6 +39,7 @@ class HTMLPurifier_AttrDef_CSSLength extends HTMLPurifier_AttrDef
        
        // we assume all units are two characters
        $unit = substr($length, $strlen - 2);
+        if (!ctype_lower($unit)) $unit = strtolower($unit);
        $number = substr($length, 0, $strlen - 2);
        
        if (!isset($this->units[$unit])) return false;
--- a/library/HTMLPurifier/AttrDef/CSS/ListStyle.php
+++ b/library/HTMLPurifier/AttrDef/CSS/ListStyle.php
@@ -0,0 +1,80 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef.php';
+
+/**
+ * Validates shorthand CSS property list-style.
+ * @warning Does not support url tokens that have internal spaces.
+ */
+class HTMLPurifier_AttrDef_CSS_ListStyle extends HTMLPurifier_AttrDef
+{
+    
+    /**
+     * Local copy of component validators.
+     * @note See HTMLPurifier_AttrDef_CSS_Font::$info for a similar impl.
+     */
+    var $info;
+    
+    function HTMLPurifier_AttrDef_CSS_ListStyle($config) {
+        $def = $config->getCSSDefinition();
+        $this->info['list-style-type']     = $def->info['list-style-type'];
+        $this->info['list-style-position'] = $def->info['list-style-position'];
+        $this->info['list-style-image'] = $def->info['list-style-image'];
+    }
+    
+    function validate($string, $config, &$context) {
+        
+        // regular pre-processing
+        $string = $this->parseCDATA($string);
+        if ($string === '') return false;
+        
+        // assumes URI doesn't have spaces in it
+        $bits = explode(' ', strtolower($string)); // bits to process
+        
+        $caught = array();
+        $caught['type']     = false;
+        $caught['position'] = false;
+        $caught['image']    = false;
+        
+        $i = 0; // number of catches
+        $none = false;
+        
+        foreach ($bits as $bit) {
+            if ($i >= 3) return; // optimization bit
+            if ($bit === '') continue;
+            foreach ($caught as $key => $status) {
+                if ($status !== false) continue;
+                $r = $this->info['list-style-' . $key]->validate($bit, $config, $context);
+                if ($r === false) continue;
+                if ($r === 'none') {
+                    if ($none) continue;
+                    else $none = true;
+                    if ($key == 'image') continue;
+                }
+                $caught[$key] = $r;
+                $i++;
+                break;
+            }
+        }
+        
+        if (!$i) return false;
+        
+        $ret = array();
+        
+        // construct type
+        if ($caught['type']) $ret[] = $caught['type'];
+        
+        // construct image
+        if ($caught['image']) $ret[] = $caught['image'];
+        
+        // construct position
+        if ($caught['position']) $ret[] = $caught['position'];
+        
+        if (empty($ret)) return false;
+        return implode(' ', $ret);
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/CSS/Multiple.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Multiple.php
@@ -13,7 +13,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 *       can only be used alone: it will never manifest as part of a multi
 *       shorthand declaration.  Thus, this class does not allow inherit.
 */
-class HTMLPurifier_AttrDef_Multiple extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Multiple extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -30,7 +30,7 @@ class HTMLPurifier_AttrDef_Multiple extends HTMLPurifier_AttrDef
     * @param $single HTMLPurifier_AttrDef to multiply
     * @param $max Max number of values allowed (usually four)
     */
-    function HTMLPurifier_AttrDef_Multiple($single, $max = 4) {
+    function HTMLPurifier_AttrDef_CSS_Multiple($single, $max = 4) {
        $this->single = $single;
        $this->max = $max;
    }
--- a/library/HTMLPurifier/AttrDef/CSS/Number.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Number.php
@@ -3,7 +3,7 @@
 /**
 * Validates a number as defined by the CSS spec.
 */
-class HTMLPurifier_AttrDef_Number extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Number extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -14,7 +14,7 @@ class HTMLPurifier_AttrDef_Number extends HTMLPurifier_AttrDef
    /**
     * @param $non_negative Bool indicating whether negatives are forbidden
     */
-    function HTMLPurifier_AttrDef_Number($non_negative = false) {
+    function HTMLPurifier_AttrDef_CSS_Number($non_negative = false) {
        $this->non_negative = $non_negative;
    }
    
--- a/library/HTMLPurifier/AttrDef/CSS/Percentage.php
+++ b/library/HTMLPurifier/AttrDef/CSS/Percentage.php
@@ -1,25 +1,24 @@
 <?php

 require_once 'HTMLPurifier/AttrDef.php';
-require_once 'HTMLPurifier/AttrDef/Number.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Number.php';

 /**
- * Validates a Percentage as defined by the HTML spec.
- * @note This also allows integer pixel values.
+ * Validates a Percentage as defined by the CSS spec.
 */
-class HTMLPurifier_AttrDef_Percentage extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_Percentage extends HTMLPurifier_AttrDef
 {
    
    /**
-     * Instance of HTMLPurifier_AttrDef_Number to defer pixel validation
+     * Instance of HTMLPurifier_AttrDef_CSS_Number to defer number validation
     */
    var $number_def;
    
    /**
     * @param Bool indicating whether to forbid negative values
     */
-    function HTMLPurifier_AttrDef_Percentage($non_negative = false) {
-        $this->number_def = new HTMLPurifier_AttrDef_Number($non_negative);
+    function HTMLPurifier_AttrDef_CSS_Percentage($non_negative = false) {
+        $this->number_def = new HTMLPurifier_AttrDef_CSS_Number($non_negative);
    }
    
    function validate($string, $config, &$context) {
--- a/library/HTMLPurifier/AttrDef/CSS/TextDecoration.php
+++ b/library/HTMLPurifier/AttrDef/CSS/TextDecoration.php
@@ -7,7 +7,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 * @note This class could be generalized into a version that acts sort of
 *       like Enum except you can compound the allowed values.
 */
-class HTMLPurifier_AttrDef_TextDecoration extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_CSS_TextDecoration extends HTMLPurifier_AttrDef
 {
    
    /**
--- a/library/HTMLPurifier/AttrDef/CSS/URI.php
+++ b/library/HTMLPurifier/AttrDef/CSS/URI.php
@@ -0,0 +1,58 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef/URI.php';
+
+/**
+ * Validates a URI in CSS syntax, which uses url('http://example.com')
+ * @note While theoretically speaking a URI in a CSS document could
+ *       be non-embedded, as of CSS2 there is no such usage so we're
+ *       generalizing it. This may need to be changed in the future.
+ * @warning Since HTMLPurifier_AttrDef_CSS blindly uses semicolons as
+ *          the separator, you cannot put a literal semicolon in
+ *          in the URI. Try percent encoding it, in that case.
+ */
+class HTMLPurifier_AttrDef_CSS_URI extends HTMLPurifier_AttrDef_URI
+{
+    
+    function HTMLPurifier_AttrDef_CSS_URI() {
+        $this->HTMLPurifier_AttrDef_URI(true); // always embedded
+    }
+    
+    function validate($uri_string, $config, &$context) {
+        // parse the URI out of the string and then pass it onto
+        // the parent object
+        
+        $uri_string = $this->parseCDATA($uri_string);
+        if (strpos($uri_string, 'url(') !== 0) return false;
+        $uri_string = substr($uri_string, 4);
+        $new_length = strlen($uri_string) - 1;
+        if ($uri_string[$new_length] != ')') return false;
+        $uri = trim(substr($uri_string, 0, $new_length));
+        
+        if (isset($uri[0]) && ($uri[0] == "'" || $uri[0] == '"')) {
+            $quote = $uri[0];
+            $new_length = strlen($uri) - 1;
+            if ($uri[$new_length] !== $quote) return false;
+            $uri = substr($uri, 1, $new_length - 1);
+        }
+        
+        $keys   = array(  '(',   ')',   ',',   ' ',   '"',   "'");
+        $values = array('\\(', '\\)', '\\,', '\\ ', '\\"', "\\'");
+        $uri = str_replace($values, $keys, $uri);
+        
+        $result = parent::validate($uri, $config, $context);
+        
+        if ($result === false) return false;
+        
+        // escape necessary characters according to CSS spec
+        // except for the comma, none of these should appear in the
+        // URI at all
+        $result = str_replace($keys, $values, $result);
+        
+        return "url($result)";
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/Enum.php
+++ b/library/HTMLPurifier/AttrDef/Enum.php
@@ -25,8 +25,8 @@ class HTMLPurifier_AttrDef_Enum extends HTMLPurifier_AttrDef
     * @param $case_sensitive Bool indicating whether or not case sensitive
     */
    function HTMLPurifier_AttrDef_Enum(
-        $valid_values = array(), $case_sensitive = false) {
-        
+        $valid_values = array(), $case_sensitive = false
+    ) {
        $this->valid_values = array_flip($valid_values);
        $this->case_sensitive = $case_sensitive;
    }
--- a/library/HTMLPurifier/AttrDef/HTML/ID.php
+++ b/library/HTMLPurifier/AttrDef/HTML/ID.php
@@ -3,6 +3,22 @@
 require_once 'HTMLPurifier/AttrDef.php';
 require_once 'HTMLPurifier/IDAccumulator.php';

+HTMLPurifier_ConfigSchema::define(
+    'Attr', 'EnableID', false, 'bool',
+    'Allows the ID attribute in HTML.  This is disabled by default '.
+    'due to the fact that without proper configuration user input can '.
+    'easily break the validation of a webpage by specifying an ID that is '.
+    'already on the surrounding HTML.  If you don\'t mind throwing caution to '.
+    'the wind, enable this directive, but I strongly recommend you also '.
+    'consider blacklisting IDs you use (%Attr.IDBlacklist) or prefixing all '.
+    'user supplied IDs (%Attr.IDPrefix).  This directive has been available '.
+    'since 1.2.0, and when set to true reverts to the behavior of pre-1.2.0 '.
+    'versions.'
+);
+HTMLPurifier_ConfigSchema::defineAlias(
+    'HTML', 'EnableAttrID', 'Attr', 'EnableID'
+);
+
 HTMLPurifier_ConfigSchema::define(
    'Attr', 'IDPrefix', '', 'string',
    'String to prefix to IDs.  If you have no idea what IDs your pages '.
@@ -27,6 +43,14 @@ HTMLPurifier_ConfigSchema::define(
    'is set to a non-empty value! This directive was available since 1.2.0.'
 );

+HTMLPurifier_ConfigSchema::define(
+    'Attr', 'IDBlacklistRegexp', null, 'string/null',
+    'PCRE regular expression to be matched against all IDs. If the expression '.
+    'is matches, the ID is rejected. Use this with care: may cause '.
+    'significant degradation. ID matching is done after all other '.
+    'validation. This directive was available since 1.6.0.'
+);
+
 /**
 * Validates the HTML attribute ID.
 * @warning Even though this is the id processor, it
@@ -36,11 +60,16 @@ HTMLPurifier_ConfigSchema::define(
 *          blacklist. If you're hacking around, make sure you use load()!
 */

-class HTMLPurifier_AttrDef_ID extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_HTML_ID extends HTMLPurifier_AttrDef
 {
    
+    // ref functionality disabled, since we also have to verify
+    // whether or not the ID it refers to exists
+    
    function validate($id, $config, &$context) {
        
+        if (!$config->get('Attr', 'EnableID')) return false;
+        
        $id = trim($id); // trim it first
        
        if ($id === '') return false;
@@ -55,8 +84,10 @@ class HTMLPurifier_AttrDef_ID extends HTMLPurifier_AttrDef
                '%Attr.IDPrefix is set', E_USER_WARNING);
        }
        
-        $id_accumulator =& $context->get('IDAccumulator');
-        if (isset($id_accumulator->ids[$id])) return false;
+        //if (!$this->ref) {
+            $id_accumulator =& $context->get('IDAccumulator');
+            if (isset($id_accumulator->ids[$id])) return false;
+        //}
        
        // we purposely avoid using regex, hopefully this is faster
        
@@ -71,7 +102,12 @@ class HTMLPurifier_AttrDef_ID extends HTMLPurifier_AttrDef
            $result = ($trim === '');
        }
        
-        if ($result) $id_accumulator->add($id);
+        $regexp = $config->get('Attr', 'IDBlacklistRegexp');
+        if ($regexp && preg_match($regexp, $id)) {
+            return false;
+        }
+        
+        if (/*!$this->ref && */$result) $id_accumulator->add($id);
        
        // if no change was made to the ID, return the result
        // else, return the new id if stripping whitespace made it
--- a/library/HTMLPurifier/AttrDef/HTML/Length.php
+++ b/library/HTMLPurifier/AttrDef/HTML/Length.php
@@ -1,18 +1,16 @@
 <?php

 require_once 'HTMLPurifier/AttrDef.php';
-require_once 'HTMLPurifier/AttrDef/Pixels.php';
+require_once 'HTMLPurifier/AttrDef/HTML/Pixels.php';

 /**
 * Validates the HTML type length (not to be confused with CSS's length).
 * 
 * This accepts integer pixels or percentages as lengths for certain
- * HTML attributes. Don't use this for CSS: that's
- * HTMLPurifier_AttrDef_CSSLength which requires prefixes and allows a lot
- * more different types.
+ * HTML attributes.
 */

-class HTMLPurifier_AttrDef_Length extends HTMLPurifier_AttrDef_Pixels
+class HTMLPurifier_AttrDef_HTML_Length extends HTMLPurifier_AttrDef_HTML_Pixels
 {
    
    function validate($string, $config, &$context) {
--- a/library/HTMLPurifier/AttrDef/HTML/LinkTypes.php
+++ b/library/HTMLPurifier/AttrDef/HTML/LinkTypes.php
@@ -0,0 +1,75 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef.php';
+
+HTMLPurifier_ConfigSchema::define(
+    'Attr', 'AllowedRel', array(), 'lookup',
+    'List of allowed forward document relationships in the rel attribute. '.
+    'Common values may be nofollow or print. By default, this is empty, '.
+    'meaning that no document relationships are allowed. This directive '.
+    'was available since 1.6.0.'
+);
+
+HTMLPurifier_ConfigSchema::define(
+    'Attr', 'AllowedRev', array(), 'lookup',
+    'List of allowed reverse document relationships in the rev attribute. '.
+    'This attribute is a bit of an edge-case; if you don\'t know what it '.
+    'is for, stay away. This directive was available since 1.6.0.'
+);
+
+/**
+ * Validates a rel/rev link attribute against a directive of allowed values
+ * @note We cannot use Enum because link types allow multiple
+ *       values.
+ * @note Assumes link types are ASCII text
+ */
+class HTMLPurifier_AttrDef_HTML_LinkTypes extends HTMLPurifier_AttrDef
+{
+    
+    /** Lookup array of attribute names to configuration name */
+    var $configLookup = array(
+        'rel' => 'AllowedRel',
+        'rev' => 'AllowedRev'
+    );
+    
+    /** Name config attribute to pull. */
+    var $name;
+    
+    function HTMLPurifier_AttrDef_HTML_LinkTypes($name) {
+        if (!isset($this->configLookup[$name])) {
+            trigger_error('Unrecognized attribute name for link '.
+                'relationship.', E_USER_ERROR);
+            return;
+        }
+        $this->name = $this->configLookup[$name];
+    }
+    
+    function validate($string, $config, &$context) {
+        
+        $allowed = $config->get('Attr', $this->name);
+        if (empty($allowed)) return false;
+        
+        $string = $this->parseCDATA($string);
+        $parts = explode(' ', $string);
+        
+        // lookup to prevent duplicates
+        $ret_lookup = array();
+        foreach ($parts as $part) {
+            $part = strtolower(trim($part));
+            if (!isset($allowed[$part])) continue;
+            $ret_lookup[$part] = true;
+        }
+        
+        if (empty($ret_lookup)) return false;
+        
+        $ret_array = array();
+        foreach ($ret_lookup as $part => $bool) $ret_array[] = $part;
+        $string = implode(' ', $ret_array);
+        
+        return $string;
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrDef/HTML/MultiLength.php
+++ b/library/HTMLPurifier/AttrDef/HTML/MultiLength.php
@@ -1,7 +1,7 @@
 <?php

 require_once 'HTMLPurifier/AttrDef.php';
-require_once 'HTMLPurifier/AttrDef/Length.php';
+require_once 'HTMLPurifier/AttrDef/HTML/Length.php';

 /**
 * Validates a MultiLength as defined by the HTML spec.
@@ -9,7 +9,7 @@ require_once 'HTMLPurifier/AttrDef/Length.php';
 * A multilength is either a integer (pixel count), a percentage, or
 * a relative number.
 */
-class HTMLPurifier_AttrDef_MultiLength extends HTMLPurifier_AttrDef_Length
+class HTMLPurifier_AttrDef_HTML_MultiLength extends HTMLPurifier_AttrDef_HTML_Length
 {
    
    function validate($string, $config, &$context) {
@@ -27,12 +27,14 @@ class HTMLPurifier_AttrDef_MultiLength extends HTMLPurifier_AttrDef_Length
        
        $int = substr($string, 0, $length - 1);
        
+        if ($int == '') return '*';
        if (!is_numeric($int)) return false;
        
        $int = (int) $int;
        
-        if ($int < 0) return '0*';
-        
+        if ($int < 0) return false;
+        if ($int == 0) return '0';
+        if ($int == 1) return '*';
        return ((string) $int) . '*';
        
    }
--- a/library/HTMLPurifier/AttrDef/HTML/Nmtokens.php
+++ b/library/HTMLPurifier/AttrDef/HTML/Nmtokens.php
@@ -4,9 +4,13 @@ require_once 'HTMLPurifier/AttrDef.php';
 require_once 'HTMLPurifier/Config.php';

 /**
- * Validates the contents of the global HTML attribute class.
+ * Validates contents based on NMTOKENS attribute type.
+ * @note The only current use for this is the class attribute in HTML
+ * @note Could have some functionality factored out into Nmtoken class
+ * @warning We cannot assume this class will be used only for 'class'
+ *          attributes. Not sure how to hook in magic behavior, then.
 */
-class HTMLPurifier_AttrDef_Class extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_HTML_Nmtokens extends HTMLPurifier_AttrDef
 {
    
    function validate($string, $config, &$context) {
@@ -31,10 +35,10 @@ class HTMLPurifier_AttrDef_Class extends HTMLPurifier_AttrDef
        
        if (empty($matches[1])) return false;
        
-        // reconstruct class string
+        // reconstruct string
        $new_string = '';
-        foreach ($matches[1] as $class_names) {
-            $new_string .= $class_names . ' ';
+        foreach ($matches[1] as $token) {
+            $new_string .= $token . ' ';
        }
        $new_string = rtrim($new_string);
        
--- a/library/HTMLPurifier/AttrDef/HTML/Pixels.php
+++ b/library/HTMLPurifier/AttrDef/HTML/Pixels.php
@@ -5,7 +5,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 /**
 * Validates an integer representation of pixels according to the HTML spec.
 */
-class HTMLPurifier_AttrDef_Pixels extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_HTML_Pixels extends HTMLPurifier_AttrDef
 {
    
    function validate($string, $config, &$context) {
--- a/library/HTMLPurifier/AttrDef/Lang.php
+++ b/library/HTMLPurifier/AttrDef/Lang.php
@@ -46,7 +46,7 @@ class HTMLPurifier_AttrDef_Lang extends HTMLPurifier_AttrDef
        
        // process second subtag : $subtags[1]
        $length = strlen($subtags[1]);
-        if ($length == 0 || $length == 1 || $length > 8 || !ctype_alnum($subtags[1])) {
+        if ($length == 0 || ($length == 1 && $subtags[1] != 'x') || $length > 8 || !ctype_alnum($subtags[1])) {
            return $new_string;
        }
        if (!ctype_lower($subtags[1])) $subtags[1] = strtolower($subtags[1]);
--- a/library/HTMLPurifier/AttrDef/ListStyle.php
+++ b/library/HTMLPurifier/AttrDef/ListStyle.php
@@ -1,78 +0,0 @@
-<?php
-
-require_once 'HTMLPurifier/AttrDef.php';
-
-/**
- * Validates shorthand CSS property list-style.
- * @note This currently does not support list-style-image, as that functionality
- *       is not implemented yet elsewhere.
- */
-class HTMLPurifier_AttrDef_ListStyle extends HTMLPurifier_AttrDef
-{
-    
-    /**
-     * Local copy of component validators.
-     * @note See HTMLPurifier_AttrDef_Font::$info for a similar impl.
-     */
-    var $info;
-    
-    function HTMLPurifier_AttrDef_ListStyle($config) {
-        $def = $config->getCSSDefinition();
-        $this->info['list-style-type']     = $def->info['list-style-type'];
-        $this->info['list-style-position'] = $def->info['list-style-position'];
-    }
-    
-    function validate($string, $config, &$context) {
-        
-        // regular pre-processing
-        $string = $this->parseCDATA($string);
-        if ($string === '') return false;
-        
-        $bits = explode(' ', strtolower($string)); // bits to process
-        
-        $caught_type = false;
-        $caught_position = false;
-        $caught_none = false; // as in keyword none, which is in all of them
-        
-        $ret = '';
-        
-        foreach ($bits as $bit) {
-            if ($caught_none && ($caught_type || $caught_position)) break;
-            if ($caught_type && $caught_position) break;
-            
-            if ($bit === '') continue;
-            
-            if ($bit === 'none') {
-                if ($caught_none) continue;
-                $caught_none = true;
-                $ret .= 'none ';
-                continue;
-            }
-            
-            // if we add anymore, roll it into a loop
-            
-            $r = $this->info['list-style-type']->validate($bit, $config, $context);
-            if ($r !== false) {
-                if ($caught_type) continue;
-                $caught_type = true;
-                $ret .= $r . ' ';
-                continue;
-            }
-            
-            $r = $this->info['list-style-position']->validate($bit, $config, $context);
-            if ($r !== false) {
-                if ($caught_position) continue;
-                $caught_position = true;
-                $ret .= $r . ' ';
-                continue;
-            }
-        }
-        
-        $ret = rtrim($ret);
-        return $ret ? $ret : false;
-        
-    }
-    
-}
-
-?>
--- a/library/HTMLPurifier/AttrDef/URI.php
+++ b/library/HTMLPurifier/AttrDef/URI.php
@@ -3,7 +3,7 @@
 require_once 'HTMLPurifier/AttrDef.php';
 require_once 'HTMLPurifier/URIScheme.php';
 require_once 'HTMLPurifier/URISchemeRegistry.php';
-require_once 'HTMLPurifier/AttrDef/Host.php';
+require_once 'HTMLPurifier/AttrDef/URI/Host.php';
 require_once 'HTMLPurifier/PercentEncoder.php';

 HTMLPurifier_ConfigSchema::define(
@@ -77,6 +77,14 @@ HTMLPurifier_ConfigSchema::define(
    'This directive has been available since 1.3.0.'
 );

+HTMLPurifier_ConfigSchema::define(
+    'URI', 'Disable', false, 'bool',
+    'Disables all URIs in all forms. Not sure why you\'d want to do that '.
+    '(after all, the Internet\'s founded on the notion of a hyperlink). '.
+    'This directive has been available since 1.3.0.'
+);
+HTMLPurifier_ConfigSchema::defineAlias('Attr', 'DisableURI', 'URI', 'Disable');
+
 /**
 * Validates a URI as defined by RFC 3986.
 * @note Scheme-specific mechanics deferred to HTMLPurifier_URIScheme
@@ -92,7 +100,7 @@ class HTMLPurifier_AttrDef_URI extends HTMLPurifier_AttrDef
     * @param $embeds_resource_resource Does the URI here result in an extra HTTP request?
     */
    function HTMLPurifier_AttrDef_URI($embeds_resource = false) {
-        $this->host = new HTMLPurifier_AttrDef_Host();
+        $this->host = new HTMLPurifier_AttrDef_URI_Host();
        $this->PercentEncoder = new HTMLPurifier_PercentEncoder();
        $this->embeds_resource = (bool) $embeds_resource;
    }
@@ -102,6 +110,8 @@ class HTMLPurifier_AttrDef_URI extends HTMLPurifier_AttrDef
        // We'll write stack-based parsers later, for now, use regexps to
        // get things working as fast as possible (irony)
        
+        if ($config->get('URI', 'Disable')) return false;
+        
        // parse as CDATA
        $uri = $this->parseCDATA($uri);
        
@@ -139,10 +149,10 @@ class HTMLPurifier_AttrDef_URI extends HTMLPurifier_AttrDef
            // no need to validate the scheme's fmt since we do that when we
            // retrieve the specific scheme object from the registry
            $scheme = ctype_lower($scheme) ? $scheme : strtolower($scheme);
-            $scheme_obj =& $registry->getScheme($scheme, $config, $context);
+            $scheme_obj = $registry->getScheme($scheme, $config, $context);
            if (!$scheme_obj) return false; // invalid scheme, clean it out
        } else {
-            $scheme_obj =& $registry->getScheme(
+            $scheme_obj = $registry->getScheme(
                $config->get('URI', 'DefaultScheme'), $config, $context
            );
        }
--- a/library/HTMLPurifier/AttrDef/URI/Email.php
+++ b/library/HTMLPurifier/AttrDef/URI/Email.php
@@ -2,7 +2,7 @@

 require_once 'HTMLPurifier/AttrDef.php';

-class HTMLPurifier_AttrDef_Email extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_URI_Email extends HTMLPurifier_AttrDef
 {
    
    /**
--- a/library/HTMLPurifier/AttrDef/URI/Email/SimpleCheck.php
+++ b/library/HTMLPurifier/AttrDef/URI/Email/SimpleCheck.php
@@ -1,12 +1,12 @@
 <?php

-require_once 'HTMLPurifier/AttrDef/Email.php';
+require_once 'HTMLPurifier/AttrDef/URI/Email.php';

 /**
 * Primitive email validation class based on the regexp found at 
 * http://www.regular-expressions.info/email.html
 */
-class HTMLPurifier_AttrDef_Email_SimpleCheck extends HTMLPurifier_AttrDef_Email
+class HTMLPurifier_AttrDef_URI_Email_SimpleCheck extends HTMLPurifier_AttrDef_URI_Email
 {
    
    function validate($string, $config, &$context) {
--- a/library/HTMLPurifier/AttrDef/URI/Host.php
+++ b/library/HTMLPurifier/AttrDef/URI/Host.php
@@ -1,28 +1,28 @@
 <?php

 require_once 'HTMLPurifier/AttrDef.php';
-require_once 'HTMLPurifier/AttrDef/IPv4.php';
-require_once 'HTMLPurifier/AttrDef/IPv6.php';
+require_once 'HTMLPurifier/AttrDef/URI/IPv4.php';
+require_once 'HTMLPurifier/AttrDef/URI/IPv6.php';

 /**
 * Validates a host according to the IPv4, IPv6 and DNS (future) specifications.
 */
-class HTMLPurifier_AttrDef_Host extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_URI_Host extends HTMLPurifier_AttrDef
 {
    
    /**
-     * Instance of HTMLPurifier_AttrDef_IPv4 sub-validator
+     * Instance of HTMLPurifier_AttrDef_URI_IPv4 sub-validator
     */
    var $ipv4;
    
    /**
-     * Instance of HTMLPurifier_AttrDef_IPv6 sub-validator
+     * Instance of HTMLPurifier_AttrDef_URI_IPv6 sub-validator
     */
    var $ipv6;
    
-    function HTMLPurifier_AttrDef_Host() {
-        $this->ipv4 = new HTMLPurifier_AttrDef_IPv4();
-        $this->ipv6 = new HTMLPurifier_AttrDef_IPv6();
+    function HTMLPurifier_AttrDef_URI_Host() {
+        $this->ipv4 = new HTMLPurifier_AttrDef_URI_IPv4();
+        $this->ipv6 = new HTMLPurifier_AttrDef_URI_IPv6();
    }
    
    function validate($string, $config, &$context) {
--- a/library/HTMLPurifier/AttrDef/URI/IPv4.php
+++ b/library/HTMLPurifier/AttrDef/URI/IPv4.php
@@ -6,7 +6,7 @@ require_once 'HTMLPurifier/AttrDef.php';
 * Validates an IPv4 address
 * @author Feyd @ forums.devnetwork.net (public domain)
 */
-class HTMLPurifier_AttrDef_IPv4 extends HTMLPurifier_AttrDef
+class HTMLPurifier_AttrDef_URI_IPv4 extends HTMLPurifier_AttrDef
 {
    
    /**
@@ -15,7 +15,7 @@ class HTMLPurifier_AttrDef_IPv4 extends HTMLPurifier_AttrDef
     */
    var $ip4;
    
-    function HTMLPurifier_AttrDef_IPv4() {
+    function HTMLPurifier_AttrDef_URI_IPv4() {
        $oct = '(?:25[0-5]|2[0-4][0-9]|1[0-9]{2}|[1-9][0-9]|[0-9])'; // 0-255
        $this->ip4 = "(?:{$oct}\\.{$oct}\\.{$oct}\\.{$oct})";
    }
--- a/library/HTMLPurifier/AttrDef/URI/IPv6.php
+++ b/library/HTMLPurifier/AttrDef/URI/IPv6.php
@@ -1,6 +1,6 @@
 <?php

-require_once 'HTMLPurifier/AttrDef/IPv4.php';
+require_once 'HTMLPurifier/AttrDef/URI/IPv4.php';

 /**
 * Validates an IPv6 address.
@@ -8,7 +8,7 @@ require_once 'HTMLPurifier/AttrDef/IPv4.php';
 * @note This function requires brackets to have been removed from address
 *       in URI.
 */
-class HTMLPurifier_AttrDef_IPv6 extends HTMLPurifier_AttrDef_IPv4
+class HTMLPurifier_AttrDef_URI_IPv6 extends HTMLPurifier_AttrDef_URI_IPv4
 {
    
    function validate($aIP, $config, &$context) {
--- a/library/HTMLPurifier/AttrTransform/BdoDir.php
+++ b/library/HTMLPurifier/AttrTransform/BdoDir.php
@@ -20,7 +20,7 @@ HTMLPurifier_ConfigSchema::defineAllowedValues(
 class HTMLPurifier_AttrTransform_BdoDir extends HTMLPurifier_AttrTransform
 {
    
-    function transform($attr, $config, $context) {
+    function transform($attr, $config, &$context) {
        if (isset($attr['dir'])) return $attr;
        $attr['dir'] = $config->get('Attr', 'DefaultTextDir');
        return $attr;
--- a/library/HTMLPurifier/AttrTransform/BgColor.php
+++ b/library/HTMLPurifier/AttrTransform/BgColor.php
@@ -0,0 +1,28 @@
+<?php
+
+require_once 'HTMLPurifier/AttrTransform.php';
+
+/**
+ * Pre-transform that changes deprecated bgcolor attribute to CSS.
+ */
+class HTMLPurifier_AttrTransform_BgColor
+extends HTMLPurifier_AttrTransform {
+
+    function transform($attr, $config, &$context) {
+        
+        if (!isset($attr['bgcolor'])) return $attr;
+        
+        $bgcolor = $attr['bgcolor'];
+        unset($attr['bgcolor']);
+        // some validation should happen here
+        
+        $attr['style'] = isset($attr['style']) ? $attr['style'] : '';
+        $attr['style'] = "background-color:$bgcolor;" . $attr['style'];
+        
+        return $attr;
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrTransform/Border.php
+++ b/library/HTMLPurifier/AttrTransform/Border.php
@@ -0,0 +1,28 @@
+<?php
+
+require_once 'HTMLPurifier/AttrTransform.php';
+
+/**
+ * Pre-transform that changes deprecated border attribute to CSS.
+ */
+class HTMLPurifier_AttrTransform_Border
+extends HTMLPurifier_AttrTransform {
+
+    function transform($attr, $config, &$context) {
+        
+        if (!isset($attr['border'])) return $attr;
+        
+        $border_width = $attr['border'];
+        unset($attr['border']);
+        // some validation should happen here
+        
+        $attr['style'] = isset($attr['style']) ? $attr['style'] : '';
+        $attr['style'] = "border:{$border_width}px solid;" . $attr['style'];
+        
+        return $attr;
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrTransform/ImgRequired.php
+++ b/library/HTMLPurifier/AttrTransform/ImgRequired.php
@@ -25,7 +25,7 @@ HTMLPurifier_ConfigSchema::define(
 class HTMLPurifier_AttrTransform_ImgRequired extends HTMLPurifier_AttrTransform
 {
    
-    function transform($attr, $config, $context) {
+    function transform($attr, $config, &$context) {
        
        $src = true;
        if (!isset($attr['src'])) {
--- a/library/HTMLPurifier/AttrTransform/Lang.php
+++ b/library/HTMLPurifier/AttrTransform/Lang.php
@@ -10,7 +10,7 @@ require_once 'HTMLPurifier/AttrTransform.php';
 class HTMLPurifier_AttrTransform_Lang extends HTMLPurifier_AttrTransform
 {
    
-    function transform($attr, $config, $context) {
+    function transform($attr, $config, &$context) {
        
        $lang     = isset($attr['lang']) ? $attr['lang'] : false;
        $xml_lang = isset($attr['xml:lang']) ? $attr['xml:lang'] : false;
--- a/library/HTMLPurifier/AttrTransform/Length.php
+++ b/library/HTMLPurifier/AttrTransform/Length.php
@@ -0,0 +1,33 @@
+<?php
+
+require_once 'HTMLPurifier/AttrTransform.php';
+
+/**
+ * Class for handling width/height length attribute transformations to CSS
+ */
+class HTMLPurifier_AttrTransform_Length extends HTMLPurifier_AttrTransform
+{
+    
+    var $name;
+    var $cssName;
+    
+    function HTMLPurifier_AttrTransform_Length($name, $css_name = null) {
+        $this->name = $name;
+        $this->cssName = $css_name ? $css_name : $name;
+    }
+    
+    function transform($attr, $config, &$context) {
+        if (!isset($attr[$this->name])) return $attr;
+        $length = $attr[$this->name];
+        unset($attr[$this->name]);
+        if(ctype_digit($length)) $length .= 'px';
+        
+        $attr['style'] = isset($attr['style']) ? $attr['style'] : '';
+        $attr['style'] = $this->cssName . ":$length;" . $attr['style'];
+        
+        return $attr;
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrTransform/Name.php
+++ b/library/HTMLPurifier/AttrTransform/Name.php
@@ -0,0 +1,31 @@
+<?php
+
+require_once 'HTMLPurifier/AttrTransform.php';
+
+/**
+ * Pre-transform that changes deprecated name attribute to ID if necessary
+ */
+class HTMLPurifier_AttrTransform_Name extends HTMLPurifier_AttrTransform
+{
+    
+    function transform($attr, $config, &$context) {
+        
+        if (!isset($attr['name'])) return $attr;
+        
+        $name = $attr['name'];
+        unset($attr['name']);
+        
+        if (isset($attr['id'])) {
+            // ID already set, discard name
+            return $attr;
+        }
+        
+        $attr['id'] = $name;
+        
+        return $attr;
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/AttrTransform/TextAlign.php
+++ b/library/HTMLPurifier/AttrTransform/TextAlign.php
@@ -6,9 +6,9 @@ require_once 'HTMLPurifier/AttrTransform.php';
 * Pre-transform that changes deprecated align attribute to text-align.
 */
 class HTMLPurifier_AttrTransform_TextAlign
-    extends HTMLPurifier_AttrTransform {
+extends HTMLPurifier_AttrTransform {

-    function transform($attr, $config, $context) {
+    function transform($attr, $config, &$context) {
        
        if (!isset($attr['align'])) return $attr;
        
--- a/library/HTMLPurifier/AttrTypes.php
+++ b/library/HTMLPurifier/AttrTypes.php
@@ -0,0 +1,41 @@
+<?php
+
+require_once 'HTMLPurifier/AttrDef/HTML/ID.php';
+require_once 'HTMLPurifier/AttrDef/HTML/Length.php';
+require_once 'HTMLPurifier/AttrDef/HTML/MultiLength.php';
+require_once 'HTMLPurifier/AttrDef/HTML/Nmtokens.php';
+require_once 'HTMLPurifier/AttrDef/HTML/Pixels.php';
+require_once 'HTMLPurifier/AttrDef/Integer.php';
+require_once 'HTMLPurifier/AttrDef/Text.php';
+require_once 'HTMLPurifier/AttrDef/URI.php';
+
+/**
+ * Provides lookup array of attribute types to HTMLPurifier_AttrDef objects
+ */
+class HTMLPurifier_AttrTypes
+{
+    /**
+     * Lookup array of attribute string identifiers to concrete implementations
+     * @public
+     */
+    var $info = array();
+    
+    /**
+     * Constructs the info array
+     */
+    function HTMLPurifier_AttrTypes() {
+        $this->info['CDATA']    = new HTMLPurifier_AttrDef_Text();
+        $this->info['ID']       = new HTMLPurifier_AttrDef_HTML_ID();
+        $this->info['Length']   = new HTMLPurifier_AttrDef_HTML_Length();
+        $this->info['MultiLength'] = new HTMLPurifier_AttrDef_HTML_MultiLength();
+        $this->info['NMTOKENS'] = new HTMLPurifier_AttrDef_HTML_Nmtokens();
+        $this->info['Pixels']   = new HTMLPurifier_AttrDef_HTML_Pixels();
+        $this->info['Text']     = new HTMLPurifier_AttrDef_Text();
+        $this->info['URI']      = new HTMLPurifier_AttrDef_URI();
+        
+        // number is really a positive integer (one or more digits)
+        $this->info['Number']   = new HTMLPurifier_AttrDef_Integer(false, false, true);
+    }
+}
+
+?>
--- a/library/HTMLPurifier/CSSDefinition.php
+++ b/library/HTMLPurifier/CSSDefinition.php
@@ -1,16 +1,19 @@
 <?php

+require_once 'HTMLPurifier/AttrDef/CSS/Background.php';
+require_once 'HTMLPurifier/AttrDef/CSS/BackgroundPosition.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Border.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Color.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Composite.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Font.php';
+require_once 'HTMLPurifier/AttrDef/CSS/FontFamily.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Length.php';
+require_once 'HTMLPurifier/AttrDef/CSS/ListStyle.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Multiple.php';
+require_once 'HTMLPurifier/AttrDef/CSS/Percentage.php';
+require_once 'HTMLPurifier/AttrDef/CSS/TextDecoration.php';
+require_once 'HTMLPurifier/AttrDef/CSS/URI.php';
 require_once 'HTMLPurifier/AttrDef/Enum.php';
-require_once 'HTMLPurifier/AttrDef/Color.php';
-require_once 'HTMLPurifier/AttrDef/Composite.php';
-require_once 'HTMLPurifier/AttrDef/CSSLength.php';
-require_once 'HTMLPurifier/AttrDef/Percentage.php';
-require_once 'HTMLPurifier/AttrDef/Multiple.php';
-require_once 'HTMLPurifier/AttrDef/TextDecoration.php';
-require_once 'HTMLPurifier/AttrDef/FontFamily.php';
-require_once 'HTMLPurifier/AttrDef/Font.php';
-require_once 'HTMLPurifier/AttrDef/Border.php';
-require_once 'HTMLPurifier/AttrDef/ListStyle.php';

 /**
 * Defines allowed CSS attributes and what their values are.
@@ -40,7 +43,7 @@ class HTMLPurifier_CSSDefinition
            array('none', 'hidden', 'dotted', 'dashed', 'solid', 'double',
            'groove', 'ridge', 'inset', 'outset'), false);
        
-        $this->info['border-style'] = new HTMLPurifier_AttrDef_Multiple($border_style);
+        $this->info['border-style'] = new HTMLPurifier_AttrDef_CSS_Multiple($border_style);
        
        $this->info['clear'] = new HTMLPurifier_AttrDef_Enum(
            array('none', 'left', 'right', 'both'), false);
@@ -51,113 +54,125 @@ class HTMLPurifier_CSSDefinition
        $this->info['font-variant'] = new HTMLPurifier_AttrDef_Enum(
            array('normal', 'small-caps'), false);
        
+        $uri_or_none = new HTMLPurifier_AttrDef_CSS_Composite(
+            array(
+                new HTMLPurifier_AttrDef_Enum(array('none')),
+                new HTMLPurifier_AttrDef_CSS_URI()
+            )
+        );
+        
        $this->info['list-style-position'] = new HTMLPurifier_AttrDef_Enum(
            array('inside', 'outside'), false);
        $this->info['list-style-type'] = new HTMLPurifier_AttrDef_Enum(
            array('disc', 'circle', 'square', 'decimal', 'lower-roman',
-            'upper-roman', 'lower-alpha', 'upper-alpha'), false);
+            'upper-roman', 'lower-alpha', 'upper-alpha', 'none'), false);
+        $this->info['list-style-image'] = $uri_or_none;
        
-        $this->info['list-style'] = new HTMLPurifier_AttrDef_ListStyle($config);
+        $this->info['list-style'] = new HTMLPurifier_AttrDef_CSS_ListStyle($config);
        
        $this->info['text-transform'] = new HTMLPurifier_AttrDef_Enum(
            array('capitalize', 'uppercase', 'lowercase', 'none'), false);
-        $this->info['color'] = new HTMLPurifier_AttrDef_Color();
+        $this->info['color'] = new HTMLPurifier_AttrDef_CSS_Color();
        
-        // technically speaking, this one should get its own validator, but
-        // since we don't support background images, it effectively is
-        // equivalent to color.  The only trouble is that if the author
-        // specifies an image and a color, they'll both end up getting dropped,
-        // even though we ought to implement it and just discard the image
-        // info.  This will be fixed in a later version (see TODO) when
-        // better URI filtering is implemented.
-        $this->info['background'] = 
+        $this->info['background-image'] = $uri_or_none;
+        $this->info['background-repeat'] = new HTMLPurifier_AttrDef_Enum(
+            array('repeat', 'repeat-x', 'repeat-y', 'no-repeat')
+        );
+        $this->info['background-attachment'] = new HTMLPurifier_AttrDef_Enum(
+            array('scroll', 'fixed')
+        );
+        $this->info['background-position'] = new HTMLPurifier_AttrDef_CSS_BackgroundPosition();
        
        $border_color = 
        $this->info['border-top-color'] = 
        $this->info['border-bottom-color'] = 
        $this->info['border-left-color'] = 
        $this->info['border-right-color'] = 
-        $this->info['background-color'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['background-color'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('transparent')),
-            new HTMLPurifier_AttrDef_Color()
+            new HTMLPurifier_AttrDef_CSS_Color()
        ));
        
-        $this->info['border-color'] = new HTMLPurifier_AttrDef_Multiple($border_color);
+        $this->info['background'] = new HTMLPurifier_AttrDef_CSS_Background($config);
+        
+        $this->info['border-color'] = new HTMLPurifier_AttrDef_CSS_Multiple($border_color);
        
        $border_width = 
        $this->info['border-top-width'] = 
        $this->info['border-bottom-width'] = 
        $this->info['border-left-width'] = 
-        $this->info['border-right-width'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['border-right-width'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('thin', 'medium', 'thick')),
-            new HTMLPurifier_AttrDef_CSSLength(true) //disallow negative
+            new HTMLPurifier_AttrDef_CSS_Length(true) //disallow negative
        ));
        
-        $this->info['border-width'] = new HTMLPurifier_AttrDef_Multiple($border_width);
+        $this->info['border-width'] = new HTMLPurifier_AttrDef_CSS_Multiple($border_width);
        
-        $this->info['letter-spacing'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['letter-spacing'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('normal')),
-            new HTMLPurifier_AttrDef_CSSLength()
+            new HTMLPurifier_AttrDef_CSS_Length()
        ));
        
-        $this->info['word-spacing'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['word-spacing'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('normal')),
-            new HTMLPurifier_AttrDef_CSSLength()
+            new HTMLPurifier_AttrDef_CSS_Length()
        ));
        
-        $this->info['font-size'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['font-size'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('xx-small', 'x-small',
                'small', 'medium', 'large', 'x-large', 'xx-large',
                'larger', 'smaller')),
-            new HTMLPurifier_AttrDef_Percentage(),
-            new HTMLPurifier_AttrDef_CSSLength()
+            new HTMLPurifier_AttrDef_CSS_Percentage(),
+            new HTMLPurifier_AttrDef_CSS_Length()
        ));
        
-        $this->info['line-height'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['line-height'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('normal')),
-            new HTMLPurifier_AttrDef_Number(true), // no negatives
-            new HTMLPurifier_AttrDef_CSSLength(true),
-            new HTMLPurifier_AttrDef_Percentage(true)
+            new HTMLPurifier_AttrDef_CSS_Number(true), // no negatives
+            new HTMLPurifier_AttrDef_CSS_Length(true),
+            new HTMLPurifier_AttrDef_CSS_Percentage(true)
        ));
        
        $margin =
        $this->info['margin-top'] = 
        $this->info['margin-bottom'] = 
        $this->info['margin-left'] = 
-        $this->info['margin-right'] = new HTMLPurifier_AttrDef_Composite(array(
-            new HTMLPurifier_AttrDef_CSSLength(),
-            new HTMLPurifier_AttrDef_Percentage(),
+        $this->info['margin-right'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
+            new HTMLPurifier_AttrDef_CSS_Length(),
+            new HTMLPurifier_AttrDef_CSS_Percentage(),
            new HTMLPurifier_AttrDef_Enum(array('auto'))
        ));
        
-        $this->info['margin'] = new HTMLPurifier_AttrDef_Multiple($margin);
+        $this->info['margin'] = new HTMLPurifier_AttrDef_CSS_Multiple($margin);
        
        // non-negative
        $padding =
        $this->info['padding-top'] = 
        $this->info['padding-bottom'] = 
        $this->info['padding-left'] = 
-        $this->info['padding-right'] = new HTMLPurifier_AttrDef_Composite(array(
-            new HTMLPurifier_AttrDef_CSSLength(true),
-            new HTMLPurifier_AttrDef_Percentage(true)
+        $this->info['padding-right'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
+            new HTMLPurifier_AttrDef_CSS_Length(true),
+            new HTMLPurifier_AttrDef_CSS_Percentage(true)
        ));
        
-        $this->info['padding'] = new HTMLPurifier_AttrDef_Multiple($padding);
+        $this->info['padding'] = new HTMLPurifier_AttrDef_CSS_Multiple($padding);
        
-        $this->info['text-indent'] = new HTMLPurifier_AttrDef_Composite(array(
-            new HTMLPurifier_AttrDef_CSSLength(),
-            new HTMLPurifier_AttrDef_Percentage()
+        $this->info['text-indent'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
+            new HTMLPurifier_AttrDef_CSS_Length(),
+            new HTMLPurifier_AttrDef_CSS_Percentage()
        ));
        
-        $this->info['width'] = new HTMLPurifier_AttrDef_Composite(array(
-            new HTMLPurifier_AttrDef_CSSLength(true),
-            new HTMLPurifier_AttrDef_Percentage(true),
+        $this->info['width'] =
+        $this->info['height'] = 
+        new HTMLPurifier_AttrDef_CSS_Composite(array(
+            new HTMLPurifier_AttrDef_CSS_Length(true),
+            new HTMLPurifier_AttrDef_CSS_Percentage(true),
            new HTMLPurifier_AttrDef_Enum(array('auto'))
        ));
        
-        $this->info['text-decoration'] = new HTMLPurifier_AttrDef_TextDecoration();
+        $this->info['text-decoration'] = new HTMLPurifier_AttrDef_CSS_TextDecoration();
        
-        $this->info['font-family'] = new HTMLPurifier_AttrDef_FontFamily();
+        $this->info['font-family'] = new HTMLPurifier_AttrDef_CSS_FontFamily();
        
        // this could use specialized code
        $this->info['font-weight'] = new HTMLPurifier_AttrDef_Enum(
@@ -166,14 +181,14 @@ class HTMLPurifier_CSSDefinition
        
        // MUST be called after other font properties, as it references
        // a CSSDefinition object
-        $this->info['font'] = new HTMLPurifier_AttrDef_Font($config);
+        $this->info['font'] = new HTMLPurifier_AttrDef_CSS_Font($config);
        
        // same here
        $this->info['border'] =
        $this->info['border-bottom'] = 
        $this->info['border-top'] = 
        $this->info['border-left'] = 
-        $this->info['border-right'] = new HTMLPurifier_AttrDef_Border($config);
+        $this->info['border-right'] = new HTMLPurifier_AttrDef_CSS_Border($config);
        
        $this->info['border-collapse'] = new HTMLPurifier_AttrDef_Enum(array(
            'collapse', 'seperate'));
@@ -184,11 +199,11 @@ class HTMLPurifier_CSSDefinition
        $this->info['table-layout'] = new HTMLPurifier_AttrDef_Enum(array(
            'auto', 'fixed'));
        
-        $this->info['vertical-align'] = new HTMLPurifier_AttrDef_Composite(array(
+        $this->info['vertical-align'] = new HTMLPurifier_AttrDef_CSS_Composite(array(
            new HTMLPurifier_AttrDef_Enum(array('baseline', 'sub', 'super',
                'top', 'text-top', 'middle', 'bottom', 'text-bottom')),
-            new HTMLPurifier_AttrDef_CSSLength(),
-            new HTMLPurifier_AttrDef_Percentage()
+            new HTMLPurifier_AttrDef_CSS_Length(),
+            new HTMLPurifier_AttrDef_CSS_Percentage()
        ));
        
    }
--- a/library/HTMLPurifier/ChildDef/Chameleon.php
+++ b/library/HTMLPurifier/ChildDef/Chameleon.php
@@ -38,22 +38,13 @@ class HTMLPurifier_ChildDef_Chameleon extends HTMLPurifier_ChildDef
    }
    
    function validateChildren($tokens_of_children, $config, &$context) {
-        $parent_type = $context->get('ParentType');
-        switch ($parent_type) {
-            case 'unknown':
-            case 'inline':
-                $result = $this->inline->validateChildren(
-                    $tokens_of_children, $config, $context);
-                break;
-            case 'block':
-                $result = $this->block->validateChildren(
-                    $tokens_of_children, $config, $context);
-                break;
-            default:
-                trigger_error('Invalid context', E_USER_ERROR);
-                return false;
+        if ($context->get('IsInline') === false) {
+            return $this->block->validateChildren(
+                $tokens_of_children, $config, $context);
+        } else {
+            return $this->inline->validateChildren(
+                $tokens_of_children, $config, $context);
        }
-        return $result;
    }
 }

--- a/library/HTMLPurifier/ChildDef/Required.php
+++ b/library/HTMLPurifier/ChildDef/Required.php
@@ -20,10 +20,13 @@ class HTMLPurifier_ChildDef_Required extends HTMLPurifier_ChildDef
            $elements = str_replace(' ', '', $elements);
            $elements = explode('|', $elements);
        }
-        $elements = array_flip($elements);
-        foreach ($elements as $i => $x) {
-            $elements[$i] = true;
-            if (empty($i)) unset($elements[$i]);
+        $keys = array_keys($elements);
+        if ($keys == array_keys($keys)) {
+            $elements = array_flip($elements);
+            foreach ($elements as $i => $x) {
+                $elements[$i] = true;
+                if (empty($i)) unset($elements[$i]);
+            }
        }
        $this->elements = $elements;
        $this->gen = new HTMLPurifier_Generator();
--- a/library/HTMLPurifier/ChildDef/StrictBlockquote.php
+++ b/library/HTMLPurifier/ChildDef/StrictBlockquote.php
@@ -4,27 +4,31 @@ require_once 'HTMLPurifier/ChildDef/Required.php';

 /**
 * Takes the contents of blockquote when in strict and reformats for validation.
- * 
- * From XHTML 1.0 Transitional to Strict, there is a notable change where 
 */
 class   HTMLPurifier_ChildDef_StrictBlockquote
 extends HTMLPurifier_ChildDef_Required
 {
+    var $real_elements;
+    var $fake_elements;
    var $allow_empty = true;
    var $type = 'strictblockquote';
    var $init = false;
-    function HTMLPurifier_ChildDef_StrictBlockquote() {}
    function validateChildren($tokens_of_children, $config, &$context) {
        
        $def = $config->getHTMLDefinition();
        if (!$this->init) {
            // allow all inline elements
-            $this->elements = $def->info_flow_elements;
-            $this->elements['#PCDATA'] = true;
+            $this->real_elements = $this->elements;
+            $this->fake_elements = $def->info_content_sets['Flow'];
+            $this->fake_elements['#PCDATA'] = true;
            $this->init = true;
        }
        
+        // trick the parent class into thinking it allows more
+        $this->elements = $this->fake_elements;
        $result = parent::validateChildren($tokens_of_children, $config, $context);
+        $this->elements = $this->real_elements;
+        
        if ($result === false) return array();
        if ($result === true) $result = $tokens_of_children;
        
@@ -40,8 +44,10 @@ extends HTMLPurifier_ChildDef_Required
            // ifs are nested for readability
            if (!$is_inline) {
                if (!$depth) {
-                     if (($token->type == 'text') ||
-                         ($def->info[$token->name]->type == 'inline')) {
+                     if (
+                        $token->type == 'text' ||
+                        !isset($this->elements[$token->name])
+                     ) {
                        $is_inline = true;
                        $ret[] = $block_wrap_start;
                     }
@@ -50,7 +56,7 @@ extends HTMLPurifier_ChildDef_Required
                if (!$depth) {
                    // starting tokens have been inline text / empty
                    if ($token->type == 'start' || $token->type == 'empty') {
-                        if ($def->info[$token->name]->type == 'block') {
+                        if (isset($this->elements[$token->name])) {
                            // ended
                            $ret[] = $block_wrap_end;
                            $is_inline = false;
--- a/library/HTMLPurifier/Config.php
+++ b/library/HTMLPurifier/Config.php
@@ -46,20 +46,24 @@ class HTMLPurifier_Config
    
    /**
     * Convenience constructor that creates a config object based on a mixed var
+     * @static
     * @param mixed $config Variable that defines the state of the config
-     *                      object. Can be: a HTMLPurifier_Config() object or
-     *                      an array of directives based on loadArray().
+     *                      object. Can be: a HTMLPurifier_Config() object,
+     *                      an array of directives based on loadArray(),
+     *                      or a string filename of an ini file.
     * @return Configured HTMLPurifier_Config object
     */
    function create($config) {
        if (is_a($config, 'HTMLPurifier_Config')) return $config;
        $ret = HTMLPurifier_Config::createDefault();
-        if (is_array($config)) $ret->loadArray($config);
+        if (is_string($config)) $ret->loadIni($config);
+        elseif (is_array($config)) $ret->loadArray($config);
        return $ret;
    }
    
    /**
     * Convenience constructor that creates a default configuration object.
+     * @static
     * @return Default HTMLPurifier_Config object.
     */
    function createDefault() {
@@ -73,12 +77,17 @@ class HTMLPurifier_Config
     * @param $namespace String namespace
     * @param $key String key
     */
-    function get($namespace, $key) {
+    function get($namespace, $key, $from_alias = false) {
        if (!isset($this->def->info[$namespace][$key])) {
            trigger_error('Cannot retrieve value of undefined directive',
                E_USER_WARNING);
            return;
        }
+        if ($this->def->info[$namespace][$key]->class == 'alias') {
+            trigger_error('Cannot get value from aliased directive, use real name',
+                E_USER_ERROR);
+            return;
+        }
        return $this->conf[$namespace][$key];
    }
    
@@ -101,12 +110,22 @@ class HTMLPurifier_Config
     * @param $key String key
     * @param $value Mixed value
     */
-    function set($namespace, $key, $value) {
+    function set($namespace, $key, $value, $from_alias = false) {
        if (!isset($this->def->info[$namespace][$key])) {
            trigger_error('Cannot set undefined directive to value',
                E_USER_WARNING);
            return;
        }
+        if ($this->def->info[$namespace][$key]->class == 'alias') {
+            if ($from_alias) {
+                trigger_error('Double-aliases not allowed, please fix '.
+                    'ConfigSchema bug');
+            }
+            $this->set($this->def->info[$namespace][$key]->namespace,
+                       $this->def->info[$namespace][$key]->name,
+                       $value, true);
+            return;
+        }
        $value = $this->def->validate(
                    $value,
                    $this->def->info[$namespace][$key]->type,
@@ -130,23 +149,36 @@ class HTMLPurifier_Config
            return;
        }
        $this->conf[$namespace][$key] = $value;
+        if ($namespace == 'HTML' || $namespace == 'Attr') {
+            // reset HTML definition if relevant attributes changed
+            $this->html_definition = null;
+        }
+        if ($namespace == 'CSS') {
+            $this->css_definition = null;
+        }
    }
    
    /**
-     * Retrieves a copy of the HTML definition.
+     * Retrieves reference to the HTML definition.
+     * @param $raw Return a copy that has not been setup yet. Must be
+     *             called before it's been setup, otherwise won't work.
     */
-    function getHTMLDefinition() {
-        if ($this->html_definition === null) {
-            $this->html_definition = new HTMLPurifier_HTMLDefinition();
-            $this->html_definition->setup($this);
+    function &getHTMLDefinition($raw = false) {
+        if (
+            empty($this->html_definition) || // hasn't ever been setup
+            ($raw && $this->html_definition->setup) // requesting new one
+        ) {
+            $this->html_definition = new HTMLPurifier_HTMLDefinition($this);
+            if ($raw) return $this->html_definition; // no setup!
        }
+        if (!$this->html_definition->setup) $this->html_definition->setup();
        return $this->html_definition;
    }
    
    /**
-     * Retrieves a copy of the CSS definition
+     * Retrieves reference to the CSS definition
     */
-    function getCSSDefinition() {
+    function &getCSSDefinition() {
        if ($this->css_definition === null) {
            $this->css_definition = new HTMLPurifier_CSSDefinition();
            $this->css_definition->setup($this);
@@ -176,6 +208,15 @@ class HTMLPurifier_Config
        }
    }
    
+    /**
+     * Loads configuration values from an ini file
+     * @param $filename Name of ini file
+     */
+    function loadIni($filename) {
+        $array = parse_ini_file($filename, true);
+        $this->loadArray($array);
+    }
+    
 }

-?>
+?>
--- a/library/HTMLPurifier/ConfigDef.php
+++ b/library/HTMLPurifier/ConfigDef.php
@@ -0,0 +1,10 @@
+<?php
+
+/**
+ * Base class for configuration entity
+ */
+class HTMLPurifier_ConfigDef {
+    var $class = false;
+}
+
+?>
--- a/library/HTMLPurifier/ConfigDef/Directive.php
+++ b/library/HTMLPurifier/ConfigDef/Directive.php
@@ -0,0 +1,74 @@
+<?php
+
+require_once 'HTMLPurifier/ConfigDef.php';
+
+/**
+ * Structure object containing definition of a directive.
+ * @note This structure does not contain default values
+ */
+class HTMLPurifier_ConfigDef_Directive extends HTMLPurifier_ConfigDef
+{
+    
+    var $class = 'directive';
+    
+    function HTMLPurifier_ConfigDef_Directive(
+        $type = null,
+        $descriptions = null,
+        $allow_null = null,
+        $allowed = null,
+        $aliases = null
+    ) {
+        if (        $type !== null)         $this->type = $type;
+        if ($descriptions !== null) $this->descriptions = $descriptions;
+        if (  $allow_null !== null)   $this->allow_null = $allow_null;
+        if (     $allowed !== null)      $this->allowed = $allowed;
+        if (     $aliases !== null)      $this->aliases = $aliases;
+    }
+    
+    /**
+     * Allowed type of the directive. Values are:
+     *      - string
+     *      - istring (case insensitive string)
+     *      - int
+     *      - float
+     *      - bool
+     *      - lookup (array of value => true)
+     *      - list (regular numbered index array)
+     *      - hash (array of key => value)
+     *      - mixed (anything goes)
+     */
+    var $type = 'mixed';
+    
+    /**
+     * Plaintext descriptions of the configuration entity is. Organized by
+     * file and line number, so multiple descriptions are allowed.
+     */
+    var $descriptions = array();
+    
+    /**
+     * Is null allowed? Has no effect for mixed type.
+     * @bool
+     */
+    var $allow_null = false;
+    
+    /**
+     * Lookup table of allowed values of the element, bool true if all allowed.
+     */
+    var $allowed = true;
+    
+    /**
+     * Hash of value aliases, i.e. values that are equivalent.
+     */
+    var $aliases = array();
+    
+    /**
+     * Adds a description to the array
+     */
+    function addDescription($file, $line, $description) {
+        if (!isset($this->descriptions[$file])) $this->descriptions[$file] = array();
+        $this->descriptions[$file][$line] = $description;
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/ConfigDef/DirectiveAlias.php
+++ b/library/HTMLPurifier/ConfigDef/DirectiveAlias.php
@@ -0,0 +1,27 @@
+<?php
+
+require_once 'HTMLPurifier/ConfigDef.php';
+
+/**
+ * Structure object describing a directive alias
+ */
+class HTMLPurifier_ConfigDef_DirectiveAlias extends HTMLPurifier_ConfigDef
+{
+    var $class = 'alias';
+    
+    /**
+     * Namespace being aliased to
+     */
+    var $namespace;
+    /**
+     * Directive being aliased to
+     */
+    var $name;
+    
+    function HTMLPurifier_ConfigDef_DirectiveAlias($namespace, $name) {
+        $this->namespace = $namespace;
+        $this->name = $name;
+    }
+}
+
+?>
--- a/library/HTMLPurifier/ConfigDef/Namespace.php
+++ b/library/HTMLPurifier/ConfigDef/Namespace.php
@@ -0,0 +1,23 @@
+<?php
+
+require_once 'HTMLPurifier/ConfigDef.php';
+
+/**
+ * Structure object describing of a namespace
+ */
+class HTMLPurifier_ConfigDef_Namespace extends HTMLPurifier_ConfigDef {
+    
+    function HTMLPurifier_ConfigDef_Namespace($description = null) {
+        $this->description = $description;
+    }
+    
+    var $class = 'namespace';
+    
+    /**
+     * String description of what kinds of directives go in this namespace.
+     */
+    var $description;
+    
+}
+
+?>
--- a/library/HTMLPurifier/ConfigSchema.php
+++ b/library/HTMLPurifier/ConfigSchema.php
@@ -1,6 +1,10 @@
 <?php

 require_once 'HTMLPurifier/Error.php';
+require_once 'HTMLPurifier/ConfigDef.php';
+require_once 'HTMLPurifier/ConfigDef/Namespace.php';
+require_once 'HTMLPurifier/ConfigDef/Directive.php';
+require_once 'HTMLPurifier/ConfigDef/DirectiveAlias.php';

 /**
 * Configuration definition, defines directives and their defaults.
@@ -67,6 +71,7 @@ class HTMLPurifier_ConfigSchema {
    
    /**
     * Retrieves an instance of the application-wide configuration definition.
+     * @static
     */
    function &instance($prototype = null) {
        static $instance;
@@ -81,6 +86,7 @@ class HTMLPurifier_ConfigSchema {
    
    /**
     * Defines a directive for configuration
+     * @static
     * @warning Will fail of directive's namespace is defined
     * @param $namespace Namespace the directive is in
     * @param $name Key of directive
@@ -104,6 +110,11 @@ class HTMLPurifier_ConfigSchema {
                E_USER_ERROR);
            return;
        }
+        if (empty($description)) {
+            trigger_error('Description must be non-empty',
+                E_USER_ERROR);
+            return;
+        }
        if (isset($def->info[$namespace][$name])) {
            if (
                $def->info[$namespace][$name]->type !== $type ||
@@ -131,7 +142,7 @@ class HTMLPurifier_ConfigSchema {
                return;
            }
            $def->info[$namespace][$name] =
-                new HTMLPurifier_ConfigEntity_Directive();
+                new HTMLPurifier_ConfigDef_Directive();
            $def->info[$namespace][$name]->type = $type;
            $def->info[$namespace][$name]->allow_null = $allow_null;
            $def->defaults[$namespace][$name]   = $default;
@@ -144,6 +155,7 @@ class HTMLPurifier_ConfigSchema {
    
    /**
     * Defines a namespace for directives to be put into.
+     * @static
     * @param $namespace Namespace's name
     * @param $description Description of the namespace
     */
@@ -158,8 +170,13 @@ class HTMLPurifier_ConfigSchema {
                E_USER_ERROR);
            return;
        }
+        if (empty($description)) {
+            trigger_error('Description must be non-empty',
+                E_USER_ERROR);
+            return;
+        }
        $def->info[$namespace] = array();
-        $def->info_namespace[$namespace] = new HTMLPurifier_ConfigEntity_Namespace();
+        $def->info_namespace[$namespace] = new HTMLPurifier_ConfigDef_Namespace();
        $def->info_namespace[$namespace]->description = $description;
        $def->defaults[$namespace] = array();
    }
@@ -169,6 +186,7 @@ class HTMLPurifier_ConfigSchema {
     * 
     * Directive value aliases are convenient for developers because it lets
     * them set a directive to several values and get the same result.
+     * @static
     * @param $namespace Directive's namespace
     * @param $name Name of Directive
     * @param $alias Name of aliased value
@@ -200,6 +218,7 @@ class HTMLPurifier_ConfigSchema {
    
    /**
     * Defines a set of allowed values for a directive.
+     * @static
     * @param $namespace Namespace of directive
     * @param $name Name of directive
     * @param $allowed_values Arraylist of allowed values
@@ -211,12 +230,66 @@ class HTMLPurifier_ConfigSchema {
                E_USER_ERROR);
            return;
        }
-        if ($def->info[$namespace][$name]->allowed === true) {
-            $def->info[$namespace][$name]->allowed = array();
+        $directive =& $def->info[$namespace][$name];
+        $type = $directive->type;
+        if ($type != 'string' && $type != 'istring') {
+            trigger_error('Cannot define allowed values for directive whose type is not string',
+                E_USER_ERROR);
+            return;
+        }
+        if ($directive->allowed === true) {
+            $directive->allowed = array();
        }
        foreach ($allowed_values as $value) {
-            $def->info[$namespace][$name]->allowed[$value] = true;
+            $directive->allowed[$value] = true;
        }
+        if ($def->defaults[$namespace][$name] !== null &&
+            !isset($directive->allowed[$def->defaults[$namespace][$name]])) {
+            trigger_error('Default value must be in allowed range of variables',
+                E_USER_ERROR);
+            $directive->allowed = true; // undo undo!
+            return;
+        }
+    }
+    
+    /**
+     * Defines a directive alias for backwards compatibility
+     * @static
+     * @param $namespace
+     * @param $name Directive that will be aliased
+     * @param $new_namespace
+     * @param $new_name Directive that the alias will be to
+     */
+    function defineAlias($namespace, $name, $new_namespace, $new_name) {
+        $def =& HTMLPurifier_ConfigSchema::instance();
+        if (!isset($def->info[$namespace])) {
+            trigger_error('Cannot define directive alias in undefined namespace',
+                E_USER_ERROR);
+            return;
+        }
+        if (!ctype_alnum($name)) {
+            trigger_error('Directive name must be alphanumeric',
+                E_USER_ERROR);
+            return;
+        }
+        if (isset($def->info[$namespace][$name])) {
+            trigger_error('Cannot define alias over directive',
+                E_USER_ERROR);
+            return;
+        }
+        if (!isset($def->info[$new_namespace][$new_name])) {
+            trigger_error('Cannot define alias to undefined directive',
+                E_USER_ERROR);
+            return;
+        }
+        if ($def->info[$new_namespace][$new_name]->class == 'alias') {
+            trigger_error('Cannot define alias to alias',
+                E_USER_ERROR);
+            return;
+        }
+        $def->info[$namespace][$name] =
+            new HTMLPurifier_ConfigDef_DirectiveAlias(
+                $new_namespace, $new_name);
    }
    
    /**
@@ -310,74 +383,4 @@ class HTMLPurifier_ConfigSchema {
    }
 }

-/**
- * Base class for configuration entity
- */
-class HTMLPurifier_ConfigEntity {}
-
-/**
- * Structure object describing of a namespace
- */
-class HTMLPurifier_ConfigEntity_Namespace extends HTMLPurifier_ConfigEntity {
-    
-    /**
-     * String description of what kinds of directives go in this namespace.
-     */
-    var $description;
-    
-}
-
-/**
- * Structure object containing definition of a directive.
- * @note This structure does not contain default values
- */
-class HTMLPurifier_ConfigEntity_Directive extends HTMLPurifier_ConfigEntity
-{
-    
-    /**
-     * Hash of value aliases, i.e. values that are equivalent.
-     */
-    var $aliases = array();
-    
-    /**
-     * Lookup table of allowed values of the element, bool true if all allowed.
-     */
-    var $allowed = true;
-    
-    /**
-     * Allowed type of the directive. Values are:
-     *      - string
-     *      - istring (case insensitive string)
-     *      - int
-     *      - float
-     *      - bool
-     *      - lookup (array of value => true)
-     *      - list (regular numbered index array)
-     *      - hash (array of key => value)
-     *      - mixed (anything goes)
-     */
-    var $type = 'mixed';
-    
-    /**
-     * Is null allowed? Has no affect for mixed type.
-     * @bool
-     */
-    var $allow_null = false;
-    
-    /**
-     * Plaintext descriptions of the configuration entity is. Organized by
-     * file and line number, so multiple descriptions are allowed.
-     */
-    var $descriptions = array();
-    
-    /**
-     * Adds a description to the array
-     */
-    function addDescription($file, $line, $description) {
-        if (!isset($this->descriptions[$file])) $this->descriptions[$file] = array();
-        $this->descriptions[$file][$line] = $description;
-    }
-    
-}
-
-?>
+?>
--- a/library/HTMLPurifier/ContentSets.php
+++ b/library/HTMLPurifier/ContentSets.php
@@ -0,0 +1,148 @@
+<?php
+
+// common defs that we'll support by default
+require_once 'HTMLPurifier/ChildDef.php';
+require_once 'HTMLPurifier/ChildDef/Empty.php';
+require_once 'HTMLPurifier/ChildDef/Required.php';
+require_once 'HTMLPurifier/ChildDef/Optional.php';
+
+class HTMLPurifier_ContentSets
+{
+    
+    /**
+     * List of content set strings (pipe seperators) indexed by name.
+     * @public
+     */
+    var $info = array();
+    
+    /**
+     * List of content set lookups (element => true) indexed by name.
+     * @note This is in HTMLPurifier_HTMLDefinition->info_content_sets
+     * @public
+     */
+    var $lookup = array();
+    
+    /**
+     * Synchronized list of defined content sets (keys of info)
+     */
+    var $keys = array();
+    /**
+     * Synchronized list of defined content values (values of info)
+     */
+    var $values = array();
+    
+    /**
+     * Merges in module's content sets, expands identifiers in the content
+     * sets and populates the keys, values and lookup member variables.
+     * @param $modules List of HTMLPurifier_HTMLModule
+     */
+    function HTMLPurifier_ContentSets($modules) {
+        if (!is_array($modules)) $modules = array($modules);
+        // populate content_sets based on module hints
+        // sorry, no way of overloading
+        foreach ($modules as $module_i => $module) {
+            foreach ($module->content_sets as $key => $value) {
+                if (isset($this->info[$key])) {
+                    // add it into the existing content set
+                    $this->info[$key] = $this->info[$key] . ' | ' . $value;
+                } else {
+                    $this->info[$key] = $value;
+                }
+            }
+        }
+        // perform content_set expansions
+        $this->keys = array_keys($this->info);
+        foreach ($this->info as $i => $set) {
+            // only performed once, so infinite recursion is not
+            // a problem
+            $this->info[$i] =
+                str_replace(
+                    $this->keys,
+                    // must be recalculated each time due to
+                    // changing substitutions
+                    array_values($this->info),
+                $set);
+        }
+        $this->values = array_values($this->info);
+        
+        // generate lookup tables
+        foreach ($this->info as $name => $set) {
+            $this->lookup[$name] = $this->convertToLookup($set);
+        }
+    }
+    
+    /**
+     * Accepts a definition; generates and assigns a ChildDef for it
+     * @param $def HTMLPurifier_ElementDef reference
+     * @param $module Module that defined the ElementDef
+     */
+    function generateChildDef(&$def, $module) {
+        if (!empty($def->child)) return; // already done!
+        $content_model = $def->content_model;
+        if (is_string($content_model)) {
+            $def->content_model = str_replace(
+                $this->keys, $this->values, $content_model);
+        }
+        $def->child = $this->getChildDef($def, $module);
+    }
+    
+    /**
+     * Instantiates a ChildDef based on content_model and content_model_type
+     * member variables in HTMLPurifier_ElementDef
+     * @note This will also defer to modules for custom HTMLPurifier_ChildDef
+     *       subclasses that need content set expansion
+     * @param $def HTMLPurifier_ElementDef to have ChildDef extracted
+     * @return HTMLPurifier_ChildDef corresponding to ElementDef
+     */
+    function getChildDef($def, $module) {
+        $value = $def->content_model;
+        if (is_object($value)) {
+            trigger_error(
+                'Literal object child definitions should be stored in '.
+                'ElementDef->child not ElementDef->content_model',
+                E_USER_NOTICE
+            );
+            return $value;
+        }
+        switch ($def->content_model_type) {
+            case 'required':
+                return new HTMLPurifier_ChildDef_Required($value);
+            case 'optional':
+                return new HTMLPurifier_ChildDef_Optional($value);
+            case 'empty':
+                return new HTMLPurifier_ChildDef_Empty();
+            case 'custom':
+                return new HTMLPurifier_ChildDef_Custom($value);
+        }
+        // defer to its module
+        $return = false;
+        if ($module->defines_child_def) { // save a func call
+            $return = $module->getChildDef($def);
+        }
+        if ($return !== false) return $return;
+        // error-out
+        trigger_error(
+            'Could not determine which ChildDef class to instantiate',
+            E_USER_ERROR
+        );
+        return false;
+    }
+    
+    /**
+     * Converts a string list of elements separated by pipes into
+     * a lookup array.
+     * @param $string List of elements
+     * @return Lookup array of elements
+     */
+    function convertToLookup($string) {
+        $array = explode('|', str_replace(' ', '', $string));
+        $ret = array();
+        foreach ($array as $i => $k) {
+            $ret[$k] = true;
+        }
+        return $ret;
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/ElementDef.php
+++ b/library/HTMLPurifier/ElementDef.php
@@ -0,0 +1,122 @@
+<?php
+
+/**
+ * Structure that stores an HTML element definition. Used by
+ * HTMLPurifier_HTMLDefinition and HTMLPurifier_HTMLModule.
+ */
+class HTMLPurifier_ElementDef
+{
+    
+    /**
+     * Does the definition work by itself, or is it created solely
+     * for the purpose of merging into another definition?
+     */
+    var $standalone = true;
+    
+    /**
+     * Associative array of attribute name to HTMLPurifier_AttrDef
+     * @note Before being processed by HTMLPurifier_AttrCollections
+     *       when modules are finalized during
+     *       HTMLPurifier_HTMLDefinition->setup(), this array may also
+     *       contain an array at index 0 that indicates which attribute
+     *       collections to load into the full array. It may also
+     *       contain string indentifiers in lieu of HTMLPurifier_AttrDef,
+     *       see HTMLPurifier_AttrTypes on how they are expanded during
+     *       HTMLPurifier_HTMLDefinition->setup() processing.
+     * @public
+     */
+    var $attr = array();
+    
+    /**
+     * Indexed list of tag's HTMLPurifier_AttrTransform to be done before validation
+     * @public
+     */
+    var $attr_transform_pre = array();
+    
+    /**
+     * Indexed list of tag's HTMLPurifier_AttrTransform to be done after validation
+     * @public
+     */
+    var $attr_transform_post = array();
+    
+    
+    
+    /**
+     * HTMLPurifier_ChildDef of this tag.
+     * @public
+     */
+    var $child;
+    
+    /**
+     * Abstract string representation of internal ChildDef rules. See
+     * HTMLPurifier_ContentSets for how this is parsed and then transformed
+     * into an HTMLPurifier_ChildDef.
+     * @public
+     */
+    var $content_model;
+    
+    /**
+     * Value of $child->type, used to determine which ChildDef to use,
+     * used in combination with $content_model.
+     * @public
+     */
+    var $content_model_type;
+    
+    
+    
+    /**
+     * Lookup table of tags that close this tag. Used during parsing
+     * to make sure we don't attempt to nest unclosed tags.
+     * @public
+     */
+    var $auto_close = array();
+    
+    /**
+     * Does the element have a content model (#PCDATA | Inline)*? This
+     * is important for chameleon ins and del processing in 
+     * HTMLPurifier_ChildDef_Chameleon. Dynamically set: modules don't
+     * have to worry about this one.
+     * @public
+     */
+    var $descendants_are_inline;
+    
+    /**
+     * Lookup table of tags excluded from all descendants of this tag.
+     * @public
+     */
+    var $excludes = array();
+    
+    /**
+     * Merges the values of another element definition into this one.
+     * Values from the new element def take precedence if a value is
+     * not mergeable.
+     */
+    function mergeIn($def) {
+        
+        // later keys takes precedence
+        foreach($def->attr as $k => $v) {
+            if ($k == 0) {
+                // merge in the includes
+                // sorry, no way to override an include
+                foreach ($v as $v2) {
+                    $def->attr[0][] = $v2;
+                }
+                continue;
+            }
+            $this->attr[$k] = $v;
+        }
+        foreach($def->attr_transform_pre    as $k => $v) $this->attr_transform_pre[$k]  = $v;
+        foreach($def->attr_transform_post   as $k => $v) $this->attr_transform_post[$k] = $v;
+        foreach($def->auto_close            as $k => $v) $this->auto_close[$k]          = $v;
+        foreach($def->excludes              as $k => $v) $this->excludes[$k]            = $v;
+        
+        if(!is_null($def->child)) $this->child = $def->child;
+        if(!empty($def->content_model)) $this->content_model .= ' | ' . $def->content_model;
+        if(!empty($def->content_model_type)) $this->content_model_type = $def->content_model_type;
+        if(!is_null($def->descendants_are_inline)) $this->descendants_are_inline = $def->descendants_are_inline;
+        
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/Encoder.php
+++ b/library/HTMLPurifier/Encoder.php
@@ -6,15 +6,29 @@ HTMLPurifier_ConfigSchema::define(
    'Core', 'Encoding', 'utf-8', 'istring', 
    'If for some reason you are unable to convert all webpages to UTF-8, '. 
    'you can use this directive as a stop-gap compatibility change to '. 
-    'let HTMLPurifier deal with non UTF-8 input.  This technique has '. 
+    'let HTML Purifier deal with non UTF-8 input.  This technique has '. 
    'notable deficiencies: absolutely no characters outside of the selected '. 
    'character encoding will be preserved, not even the ones that have '. 
    'been ampersand escaped (this is due to a UTF-8 specific <em>feature</em> '.
    'that automatically resolves all entities), making it pretty useless '.
-    'for anything except the most I18N-blind applications.  This directive '.
+    'for anything except the most I18N-blind applications, although '.
+    '%Core.EscapeNonASCIICharacters offers fixes this trouble with '.
+    'another tradeoff. This directive '.
    'only accepts ISO-8859-1 if iconv is not enabled.'
 );

+HTMLPurifier_ConfigSchema::define(
+    'Core', 'EscapeNonASCIICharacters', false, 'bool',
+    'This directive overcomes a deficiency in %Core.Encoding by blindly '.
+    'converting all non-ASCII characters into decimal numeric entities before '.
+    'converting it to its native encoding. This means that even '.
+    'characters that can be expressed in the non-UTF-8 encoding will '.
+    'be entity-ized, which can be a real downer for encodings like Big5. '.
+    'It also assumes that the ASCII repetoire is available, although '.
+    'this is the case for almost all encodings. Anyway, use UTF-8! This '.
+    'directive has been available since 1.4.0.'
+);
+
 if ( !function_exists('iconv') ) {
    // only encodings with native PHP support
    HTMLPurifier_ConfigSchema::defineAllowedValues(
@@ -38,16 +52,25 @@ HTMLPurifier_ConfigSchema::define(

 /**
 * A UTF-8 specific character encoder that handles cleaning and transforming.
+ * @note All functions in this class should be static.
 */
 class HTMLPurifier_Encoder
 {
    
+    /**
+     * Constructor throws fatal error if you attempt to instantiate class
+     */
+    function HTMLPurifier_Encoder() {
+        trigger_error('Cannot instantiate encoder, call methods statically', E_USER_ERROR);
+    }
+    
    /**
     * Cleans a UTF-8 string for well-formedness and SGML validity
     * 
     * It will parse according to UTF-8 and return a valid UTF8 string, with
     * non-SGML codepoints excluded.
     * 
+     * @static
     * @note Just for reference, the non-SGML code points are 0 to 31 and
     *       127 to 159, inclusive.  However, we allow code points 9, 10
     *       and 13, which are the tab, line feed and carriage return
@@ -225,6 +248,7 @@ class HTMLPurifier_Encoder
    
    /**
     * Translates a Unicode codepoint into its corresponding UTF-8 character.
+     * @static
     * @note Based on Feyd's function at
     *       <http://forums.devnetwork.net/viewtopic.php?p=191404#191404>,
     *       which is in public domain.
@@ -288,6 +312,7 @@ class HTMLPurifier_Encoder
    
    /**
     * Converts a string to UTF-8 based on configuration.
+     * @static
     */
    function convertToUTF8($str, $config, &$context) {
        static $iconv = null;
@@ -299,10 +324,12 @@ class HTMLPurifier_Encoder
        } elseif ($encoding === 'iso-8859-1') {
            return @utf8_encode($str);
        }
+        trigger_error('Encoding not supported', E_USER_ERROR);
    }
    
    /**
     * Converts a string from UTF-8 based on configuration.
+     * @static
     * @note Currently, this is a lossy conversion, with unexpressable
     *       characters being omitted.
     */
@@ -311,11 +338,63 @@ class HTMLPurifier_Encoder
        if ($iconv === null) $iconv = function_exists('iconv');
        $encoding = $config->get('Core', 'Encoding');
        if ($encoding === 'utf-8') return $str;
+        if ($config->get('Core', 'EscapeNonASCIICharacters')) {
+            $str = HTMLPurifier_Encoder::convertToASCIIDumbLossless($str);
+        }
        if ($iconv && !$config->get('Test', 'ForceNoIconv')) {
            return @iconv('utf-8', $encoding . '//IGNORE', $str);
        } elseif ($encoding === 'iso-8859-1') {
            return @utf8_decode($str);
        }
+        trigger_error('Encoding not supported', E_USER_ERROR);
+    }
+    
+    /**
+     * Lossless (character-wise) conversion of HTML to ASCII
+     * @static
+     * @param $str UTF-8 string to be converted to ASCII
+     * @returns ASCII encoded string with non-ASCII character entity-ized
+     * @warning Adapted from MediaWiki, claiming fair use: this is a common
+     *       algorithm. If you disagree with this license fudgery,
+     *       implement it yourself.
+     * @note Uses decimal numeric entities since they are best supported.
+     * @note This is a DUMB function: it has no concept of keeping
+     *       character entities that the projected character encoding
+     *       can allow. We could possibly implement a smart version
+     *       but that would require it to also know which Unicode
+     *       codepoints the charset supported (not an easy task).
+     * @note Sort of with cleanUTF8() but it assumes that $str is
+     *       well-formed UTF-8
+     */
+    function convertToASCIIDumbLossless($str) {
+        $bytesleft = 0;
+        $result = '';
+        $working = 0;
+        $len = strlen($str);
+        for( $i = 0; $i < $len; $i++ ) {
+            $bytevalue = ord( $str[$i] );
+            if( $bytevalue <= 0x7F ) { //0xxx xxxx
+                $result .= chr( $bytevalue );
+                $bytesleft = 0;
+            } elseif( $bytevalue <= 0xBF ) { //10xx xxxx
+                $working = $working << 6;
+                $working += ($bytevalue & 0x3F);
+                $bytesleft--;
+                if( $bytesleft <= 0 ) {
+                    $result .= "&#" . $working . ";";
+                }
+            } elseif( $bytevalue <= 0xDF ) { //110x xxxx
+                $working = $bytevalue & 0x1F;
+                $bytesleft = 1;
+            } elseif( $bytevalue <= 0xEF ) { //1110 xxxx
+                $working = $bytevalue & 0x0F;
+                $bytesleft = 2;
+            } else { //1111 0xxx
+                $working = $bytevalue & 0x07;
+                $bytesleft = 3;
+            }
+        }
+        return $result;
    }
    
    
--- a/library/HTMLPurifier/EntityLookup.php
+++ b/library/HTMLPurifier/EntityLookup.php
@@ -26,6 +26,7 @@ class HTMLPurifier_EntityLookup {
    
    /**
     * Retrieves sole instance of the object.
+     * @static
     * @param Optional prototype of custom lookup table to overload with.
     */
    function instance($prototype = false) {
--- a/library/HTMLPurifier/Filter.php
+++ b/library/HTMLPurifier/Filter.php
@@ -0,0 +1,39 @@
+<?php
+
+/**
+ * Represents a pre or post processing filter on HTML Purifier's output
+ * 
+ * Sometimes, a little ad-hoc fixing of HTML has to be done before
+ * it gets sent through HTML Purifier: you can use filters to acheive
+ * this effect. For instance, YouTube videos can be preserved using
+ * this manner. You could have used a decorator for this task, but
+ * PHP's support for them is not terribly robust, so we're going
+ * to just loop through the filters.
+ * 
+ * Filters should be exited first in, last out. If there are three filters,
+ * named 1, 2 and 3, the order of execution should go 1->preFilter,
+ * 2->preFilter, 3->preFilter, purify, 3->postFilter, 2->postFilter,
+ * 1->postFilter.
+ */
+
+class HTMLPurifier_Filter
+{
+    
+    /**
+     * Name of the filter for identification purposes
+     */
+    var $name;
+    
+    /**
+     * Pre-processor function, handles HTML before HTML Purifier 
+     */
+    function preFilter($html, $config, &$context) {}
+    
+    /**
+     * Post-processor function, handles HTML after HTML Purifier
+     */
+    function postFilter($html, $config, &$context) {}
+    
+}
+
+?>
--- a/library/HTMLPurifier/Filter/YouTube.php
+++ b/library/HTMLPurifier/Filter/YouTube.php
@@ -0,0 +1,34 @@
+<?php
+
+require_once 'HTMLPurifier/Filter.php';
+
+class HTMLPurifier_Filter_YouTube extends HTMLPurifier_Filter
+{
+    
+    var $name = 'YouTube preservation';
+    
+    function preFilter($html, $config, &$context) {
+        $pre_regex = '#<object[^>]+>.+?'.
+            'http://www.youtube.com/v/([A-Za-z0-9\-_]+).+?</object>#s';
+        $pre_replace = '<span class="youtube-embed">\1</span>';
+        return preg_replace($pre_regex, $pre_replace, $html);
+    }
+    
+    function postFilter($html, $config, &$context) {
+        $post_regex = '#<span class="youtube-embed">([A-Za-z0-9\-_]+)</span>#';
+        $post_replace = '<object width="425" height="350" '.
+            'data="http://www.youtube.com/v/\1">'.
+            '<param name="movie" value="http://www.youtube.com/v/\1"></param>'.
+            '<param name="wmode" value="transparent"></param>'.
+            '<!--[if IE]>'.
+            '<embed src="http://www.youtube.com/v/\1"'.
+            'type="application/x-shockwave-flash"'.
+            'wmode="transparent" width="425" height="350" />'.
+            '<![endif]-->'.
+            '</object>';
+        return preg_replace($post_regex, $post_replace, $html);
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/HTMLDefinition.php
+++ b/library/HTMLPurifier/HTMLDefinition.php
@@ -1,651 +1,281 @@
-<?php
-
-require_once 'HTMLPurifier/AttrDef.php';
-    require_once 'HTMLPurifier/AttrDef/Enum.php';
-    require_once 'HTMLPurifier/AttrDef/ID.php';
-    require_once 'HTMLPurifier/AttrDef/Class.php';
-    require_once 'HTMLPurifier/AttrDef/Text.php';
-    require_once 'HTMLPurifier/AttrDef/Lang.php';
-    require_once 'HTMLPurifier/AttrDef/Pixels.php';
-    require_once 'HTMLPurifier/AttrDef/Length.php';
-    require_once 'HTMLPurifier/AttrDef/MultiLength.php';
-    require_once 'HTMLPurifier/AttrDef/Integer.php';
-    require_once 'HTMLPurifier/AttrDef/URI.php';
-    require_once 'HTMLPurifier/AttrDef/CSS.php';
-require_once 'HTMLPurifier/AttrTransform.php';
-    require_once 'HTMLPurifier/AttrTransform/Lang.php';
-    require_once 'HTMLPurifier/AttrTransform/TextAlign.php';
-    require_once 'HTMLPurifier/AttrTransform/BdoDir.php';
-    require_once 'HTMLPurifier/AttrTransform/ImgRequired.php';
-require_once 'HTMLPurifier/ChildDef.php';
-    require_once 'HTMLPurifier/ChildDef/Chameleon.php';
-    require_once 'HTMLPurifier/ChildDef/Empty.php';
-    require_once 'HTMLPurifier/ChildDef/Required.php';
-    require_once 'HTMLPurifier/ChildDef/Optional.php';
-    require_once 'HTMLPurifier/ChildDef/Table.php';
-    require_once 'HTMLPurifier/ChildDef/StrictBlockquote.php';
-require_once 'HTMLPurifier/Generator.php';
-require_once 'HTMLPurifier/Token.php';
-require_once 'HTMLPurifier/TagTransform.php';
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'EnableAttrID', false, 'bool',
-    'Allows the ID attribute in HTML.  This is disabled by default '.
-    'due to the fact that without proper configuration user input can '.
-    'easily break the validation of a webpage by specifying an ID that is '.
-    'already on the surrounding HTML.  If you don\'t mind throwing caution to '.
-    'the wind, enable this directive, but I strongly recommend you also '.
-    'consider blacklisting IDs you use (%Attr.IDBlacklist) or prefixing all '.
-    'user supplied IDs (%Attr.IDPrefix).  This directive has been available '.
-    'since 1.2.0, and when set to true reverts to the behavior of pre-1.2.0 '.
-    'versions.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'Strict', false, 'bool',
-    'Determines whether or not to use Transitional (loose) or Strict rulesets. '.
-    'This directive has been available since 1.3.0.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'BlockWrapper', 'p', 'string',
-    'String name of element to wrap inline elements that are inside a block '.
-    'context.  This only occurs in the children of blockquote in strict mode. '.
-    'Example: by default value, <code>&lt;blockquote&gt;Foo&lt;/blockquote&gt;</code> '.
-    'would become <code>&lt;blockquote&gt;&lt;p&gt;Foo&lt;/p&gt;&lt;/blockquote&gt;</code>. The '.
-    '<code>&lt;p&gt;</code> tags can be replaced '.
-    'with whatever you desire, as long as it is a block level element. '.
-    'This directive has been available since 1.3.0.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'Parent', 'div', 'string',
-    'String name of element that HTML fragment passed to library will be '.
-    'inserted in.  An interesting variation would be using span as the '.
-    'parent element, meaning that only inline tags would be allowed. '.
-    'This directive has been available since 1.3.0.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'AllowedElements', null, 'lookup/null',
-    'If HTML Purifier\'s tag set is unsatisfactory for your needs, you '.
-    'can overload it with your own list of tags to allow.  Note that this '.
-    'method is subtractive: it does its job by taking away from HTML Purifier '.
-    'usual feature set, so you cannot add a tag that HTML Purifier never '.
-    'supported in the first place (like embed, form or head).  If you change this, you '.
-    'probably also want to change %HTML.AllowedAttributes. '.
-    '<strong>Warning:</strong> If another directive conflicts with the '.
-    'elements here, <em>that</em> directive will win and override. '.
-    'This directive has been available since 1.3.0.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'HTML', 'AllowedAttributes', null, 'lookup/null',
-    'IF HTML Purifier\'s attribute set is unsatisfactory, overload it! '.
-    'The syntax is \'tag.attr\' or \'*.attr\' for the global attributes '.
-    '(style, id, class, dir, lang, xml:lang).'.
-    '<strong>Warning:</strong> If another directive conflicts with the '.
-    'elements here, <em>that</em> directive will win and override. For '.
-    'example, %HTML.EnableAttrID will take precedence over *.id in this '.
-    'directive.  You must set that directive to true before you can use '.
-    'IDs at all. This directive has been available since 1.3.0.'
-);
-
-HTMLPurifier_ConfigSchema::define(
-    'Attr', 'DisableURI', false, 'bool',
-    'Disables all URIs in all forms. Not sure why you\'d want to do that '.
-    '(after all, the Internet\'s founded on the notion of a hyperlink). '.
-    'This directive has been available since 1.3.0.'
-);
-
-/**
- * Defines the purified HTML type with large amounts of objects.
- * 
- * The main function of this object is its $info array, which is an 
- * associative array of all the child and attribute definitions for
- * each allowed element. It also contains special use information (always
- * prefixed by info) for intelligent tag closing and global attributes.
- * 
- * For optimization, the definition generation may be moved to
- * a maintenance script and stipulate that definition be created
- * by a factory method that unserializes a serialized version of Definition.
- * Customization would entail copying the maintenance script, making the
- * necessary changes, generating the serialized object, and then hooking it
- * in via the factory method. We would also offer a LiveDefinition for
- * automatic recompilation, suggesting that we would have a DefinitionGenerator.
- */
-
-class HTMLPurifier_HTMLDefinition
-{
-    
-    /**
-     * Associative array of element names to HTMLPurifier_ElementDef
-     * @public
-     */
-    var $info = array();
-    
-    /**
-     * Associative array of global attribute name to attribute definition.
-     * @public
-     */
-    var $info_global_attr = array();
-    
-    /**
-     * String name of parent element HTML will be going into.
-     * @public
-     */
-    var $info_parent = 'div';
-    
-    /**
-     * Definition for parent element, allows parent element to be a
-     * tag that's not allowed inside the HTML fragment.
-     * @public
-     */
-    var $info_parent_def;
-    
-    /**
-     * String name of element used to wrap inline elements in block context
-     * @note This is rarely used except for BLOCKQUOTEs in strict mode
-     * @public
-     */
-    var $info_block_wrapper = 'p';
-    
-    /**
-     * Associative array of deprecated tag name to HTMLPurifier_TagTransform
-     * @public
-     */
-    var $info_tag_transform = array();
-    
-    /**
-     * List of HTMLPurifier_AttrTransform to be performed before validation.
-     * @public
-     */
-    var $info_attr_transform_pre = array();
-    
-    /**
-     * List of HTMLPurifier_AttrTransform to be performed after validation/
-     * @public
-     */
-    var $info_attr_transform_post = array();
-    
-    /**
-     * Lookup table of flow elements
-     * @public
-     */
-    var $info_flow_elements = array();
-    
-    /**
-     * Boolean is a strict definition?
-     * @public
-     */
-    var $strict;
-    
-    /**
-     * Initializes the definition, the meat of the class.
-     */
-    function setup($config) {
-        
-        // some cached config values
-        $this->strict = $config->get('HTML', 'Strict');
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[] : initializes the definition objects
-        
-        // if you attempt to define rules later on for a tag not in this array
-        // PHP will create an stdclass
-        
-        $allowed_tags =
-            array(
-                'ins', 'del', 'blockquote', 'dd', 'li', 'div', 'em', 'strong',
-                'dfn', 'code', 'samp', 'kbd', 'var', 'cite', 'abbr', 'acronym',
-                'q', 'sub', 'tt', 'sup', 'i', 'b', 'big', 'small',
-                'bdo', 'span', 'dt', 'p', 'h1', 'h2', 'h3', 'h4',
-                'h5', 'h6', 'ol', 'ul', 'dl', 'address', 'img', 'br', 'hr',
-                'pre', 'a', 'table', 'caption', 'thead', 'tfoot', 'tbody',
-                'colgroup', 'col', 'td', 'th', 'tr'
-            );
-        
-        if (!$this->strict) {
-            $allowed_tags[] = 'u';
-            $allowed_tags[] = 's';
-            $allowed_tags[] = 'strike';
-        }
-        
-        foreach ($allowed_tags as $tag) {
-            $this->info[$tag] = new HTMLPurifier_ElementDef();
-        }
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->child : defines allowed children for elements
-        
-        // emulates the structure of the DTD
-        // however, these are condensed, with bad stuff taken out
-        // screening process was done by hand
-        
-        // entities: prefixed with e_ and _ replaces . from DTD
-        // double underlines are entities we made up
-        
-        // we don't use an array because that complicates interpolation
-        // strings are used instead of arrays because if you use arrays,
-        // you have to do some hideous manipulation with array_merge()
-        
-        // todo: determine whether or not having allowed children
-        //       that aren't allowed globally affects security (it shouldn't)
-        // if above works out, extend children definitions to include all
-        //       possible elements (allowed elements will dictate which ones
-        //       get dropped
-        
-        $e_special_extra = 'img';
-        $e_special_basic = 'br | span | bdo';
-        $e_special = "$e_special_basic | $e_special_extra";
-        $e_fontstyle_extra = 'big | small';
-        $e_fontstyle_basic = 'tt | i | b | u | s | strike';
-        $e_fontstyle = "$e_fontstyle_basic | $e_fontstyle_extra";
-        $e_phrase_extra = 'sub | sup';
-        $e_phrase_basic = 'em | strong | dfn | code | q | samp | kbd | var'.
-          ' | cite | abbr | acronym';
-        $e_phrase = "$e_phrase_basic | $e_phrase_extra";
-        $e_misc_inline = 'ins | del';
-        $e_misc = "$e_misc_inline";
-        $e_inline = "a | $e_special | $e_fontstyle | $e_phrase";
-        // pseudo-property we created for convenience, see later on
-        $e__inline = "#PCDATA | $e_inline | $e_misc_inline";
-        // note the casing
-        $e_Inline = new HTMLPurifier_ChildDef_Optional($e__inline);
-        $e_heading = 'h1|h2|h3|h4|h5|h6';
-        $e_lists = 'ul | ol | dl';
-        $e_blocktext = 'pre | hr | blockquote | address';
-        $e_block = "p | $e_heading | div | $e_lists | $e_blocktext | table";
-        $e_Block = new HTMLPurifier_ChildDef_Optional($e_block);
-        $e__flow = "#PCDATA | $e_block | $e_inline | $e_misc";
-        $e_Flow = new HTMLPurifier_ChildDef_Optional($e__flow);
-        $e_a_content = new HTMLPurifier_ChildDef_Optional("#PCDATA".
-          " | $e_special | $e_fontstyle | $e_phrase | $e_misc_inline");
-        $e_pre_content = new HTMLPurifier_ChildDef_Optional("#PCDATA | a".
-          " | $e_special_basic | $e_fontstyle_basic | $e_phrase_basic".
-          " | $e_misc_inline");
-        $e_form_content = new HTMLPurifier_ChildDef_Optional('');//unused
-        $e_form_button_content = new HTMLPurifier_ChildDef_Optional('');//unused
-        
-        $this->info['ins']->child =
-        $this->info['del']->child =
-            new HTMLPurifier_ChildDef_Chameleon($e__inline, $e__flow);
-        
-        $this->info['dd']->child  =
-        $this->info['li']->child  =
-        $this->info['div']->child = $e_Flow;
-        
-        if ($this->strict) {
-            $this->info['blockquote']->child = new HTMLPurifier_ChildDef_StrictBlockquote();
-        } else {
-            $this->info['blockquote']->child = $e_Flow;
-        }
-        
-        $this->info['caption']->child   = 
-        $this->info['em']->child   =
-        $this->info['strong']->child    =
-        $this->info['dfn']->child  =
-        $this->info['code']->child =
-        $this->info['samp']->child =
-        $this->info['kbd']->child  =
-        $this->info['var']->child  =
-        $this->info['cite']->child =
-        $this->info['abbr']->child =
-        $this->info['acronym']->child   =
-        $this->info['q']->child    =
-        $this->info['sub']->child  =
-        $this->info['tt']->child   =
-        $this->info['sup']->child  =
-        $this->info['i']->child    =
-        $this->info['b']->child    =
-        $this->info['big']->child  =
-        $this->info['small']->child=
-        $this->info['u']->child    =
-        $this->info['s']->child    =
-        $this->info['strike']->child    =
-        $this->info['bdo']->child  =
-        $this->info['span']->child =
-        $this->info['dt']->child   =
-        $this->info['p']->child    = 
-        $this->info['h1']->child   = 
-        $this->info['h2']->child   = 
-        $this->info['h3']->child   = 
-        $this->info['h4']->child   = 
-        $this->info['h5']->child   = 
-        $this->info['h6']->child   = $e_Inline;
-        
-        // the only three required definitions, besides custom table code
-        $this->info['ol']->child   =
-        $this->info['ul']->child   = new HTMLPurifier_ChildDef_Required('li');
-        
-        $this->info['dl']->child   = new HTMLPurifier_ChildDef_Required('dt|dd');
-        
-        if ($this->strict) {
-            $this->info['address']->child = $e_Inline;
-        } else {
-            $this->info['address']->child =
-              new HTMLPurifier_ChildDef_Optional("#PCDATA | p | $e_inline".
-                  " | $e_misc_inline");
-        }
-        
-        $this->info['img']->child  =
-        $this->info['br']->child   =
-        $this->info['hr']->child   = new HTMLPurifier_ChildDef_Empty();
-        
-        $this->info['pre']->child  = $e_pre_content;
-        
-        $this->info['a']->child    = $e_a_content;
-        
-        $this->info['table']->child = new HTMLPurifier_ChildDef_Table();
-        
-        // not a real entity, watch the double underscore
-        $e__row = new HTMLPurifier_ChildDef_Required('tr');
-        $this->info['thead']->child = $e__row;
-        $this->info['tfoot']->child = $e__row;
-        $this->info['tbody']->child = $e__row;
-        $this->info['colgroup']->child = new HTMLPurifier_ChildDef_Optional('col');
-        $this->info['col']->child = new HTMLPurifier_ChildDef_Empty();
-        $this->info['tr']->child = new HTMLPurifier_ChildDef_Required('th | td');
-        $this->info['th']->child = $e_Flow;
-        $this->info['td']->child = $e_Flow;
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->type : defines the type of the element (block or inline)
-        
-        // reuses $e_Inline and $e_Block
-        foreach ($e_Inline->elements as $name => $bool) {
-            if ($name == '#PCDATA') continue;
-            $this->info[$name]->type = 'inline';
-        }
-        
-        foreach ($e_Block->elements as $name => $bool) {
-            $this->info[$name]->type = 'block';
-        }
-        
-        foreach ($e_Flow->elements as $name => $bool) {
-            $this->info_flow_elements[$name] = true;
-        }
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->excludes : defines elements that aren't allowed in here
-        
-        // make sure you test using isset() and not !empty()
-        
-        $this->info['a']->excludes = array('a' => true);
-        $this->info['pre']->excludes = array_flip(array('img', 'big', 'small',
-            // technically useless, but good to be indepth
-            'object', 'applet', 'font', 'basefont'));
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->attr : defines allowed attributes for elements
-        
-        // this doesn't include REQUIRED declarations, those are handled
-        // by the transform classes. It will, however, do simple and slightly
-        // complex attribute value substitution
-        
-        // the question of varying allowed attributes is more entangling.
-        
-        $e_Text = new HTMLPurifier_AttrDef_Text();
-        
-        // attrs, included in almost every single one except for a few,
-        // which manually override these in their local definitions
-        $this->info_global_attr = array(
-            // core attrs
-            'class' => new HTMLPurifier_AttrDef_Class(),
-            'title' => $e_Text,
-            'style' => new HTMLPurifier_AttrDef_CSS(),
-            // i18n
-            'dir'   => new HTMLPurifier_AttrDef_Enum(array('ltr','rtl'), false),
-            'lang'  => new HTMLPurifier_AttrDef_Lang(),
-            'xml:lang' => new HTMLPurifier_AttrDef_Lang(),
-            );
-        
-        if ($config->get('HTML', 'EnableAttrID')) {
-            $this->info_global_attr['id'] = new HTMLPurifier_AttrDef_ID();
-        }
-        
-        // required attribute stipulation handled in attribute transformation
-        $this->info['bdo']->attr = array(); // nothing else
-        
-        $this->info['br']->attr['dir'] = false;
-        $this->info['br']->attr['lang'] = false;
-        $this->info['br']->attr['xml:lang'] = false;
-        
-        $this->info['td']->attr['abbr'] = $e_Text;
-        $this->info['th']->attr['abbr'] = $e_Text;
-        
-        $this->setAttrForTableElements('align', new HTMLPurifier_AttrDef_Enum(
-            array('left', 'center', 'right', 'justify', 'char'), false));
-        
-        $this->setAttrForTableElements('valign', new HTMLPurifier_AttrDef_Enum(
-            array('top', 'middle', 'bottom', 'baseline'), false));
-        
-        $this->info['img']->attr['alt'] = $e_Text;
-        
-        $e_TFrame = new HTMLPurifier_AttrDef_Enum(array('void', 'above',
-            'below', 'hsides', 'lhs', 'rhs', 'vsides', 'box', 'border'), false);
-        $this->info['table']->attr['frame'] = $e_TFrame;
-        
-        $e_TRules = new HTMLPurifier_AttrDef_Enum(array('none', 'groups',
-            'rows', 'cols', 'all'), false);
-        $this->info['table']->attr['rules'] = $e_TRules;
-        
-        $this->info['table']->attr['summary'] = $e_Text;
-        
-        $this->info['table']->attr['border'] =
-            new HTMLPurifier_AttrDef_Pixels();
-        
-        $e_Length = new HTMLPurifier_AttrDef_Length();
-        $this->info['table']->attr['cellpadding'] =
-        $this->info['table']->attr['cellspacing'] =
-        $this->info['table']->attr['width'] =
-        $this->info['img']->attr['height'] =
-        $this->info['img']->attr['width'] = $e_Length;
-        $this->setAttrForTableElements('charoff', $e_Length);
-        
-        $e_MultiLength = new HTMLPurifier_AttrDef_MultiLength();
-        $this->info['col']->attr['width'] =
-        $this->info['colgroup']->attr['width'] = $e_MultiLength;
-        
-        $e__NumberSpan = new HTMLPurifier_AttrDef_Integer(false, false, true);
-        $this->info['colgroup']->attr['span'] =
-        $this->info['col']->attr['span']   =
-        $this->info['td']->attr['rowspan'] =
-        $this->info['th']->attr['rowspan'] = 
-        $this->info['td']->attr['colspan'] =
-        $this->info['th']->attr['colspan'] = $e__NumberSpan;
-        
-        if (!$config->get('Attr', 'DisableURI')) {
-            $e_URI = new HTMLPurifier_AttrDef_URI();
-            $this->info['a']->attr['href'] =
-            $this->info['img']->attr['longdesc'] =
-            $this->info['del']->attr['cite'] =
-            $this->info['ins']->attr['cite'] =
-            $this->info['blockquote']->attr['cite'] =
-            $this->info['q']->attr['cite'] = $e_URI;
-            
-            // URI that causes HTTP request
-            $this->info['img']->attr['src'] = new HTMLPurifier_AttrDef_URI(true);
-        }
-        
-        if (!$this->strict) {
-            $this->info['li']->attr['value'] = new HTMLPurifier_AttrDef_Integer();
-            $this->info['ol']->attr['start'] = new HTMLPurifier_AttrDef_Integer();
-        }
-        
-        //////////////////////////////////////////////////////////////////////
-        // info_tag_transform : transformations of tags
-        
-        $this->info_tag_transform['font']   = new HTMLPurifier_TagTransform_Font();
-        $this->info_tag_transform['menu']   = new HTMLPurifier_TagTransform_Simple('ul');
-        $this->info_tag_transform['dir']    = new HTMLPurifier_TagTransform_Simple('ul');
-        $this->info_tag_transform['center'] = new HTMLPurifier_TagTransform_Center();
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->auto_close : tags that automatically close another
-        
-        // todo: determine whether or not SGML-like modeling based on
-        // mandatory/optional end tags would be a better policy
-        
-        // make sure you test using isset() not !empty()
-        
-        // these are all block elements: blocks aren't allowed in P
-        $this->info['p']->auto_close = array_flip(array(
-                'address', 'blockquote', 'dd', 'dir', 'div', 'dl', 'dt',
-                'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'hr', 'ol', 'p', 'pre',
-                'table', 'ul'
-            ));
-        
-        $this->info['li']->auto_close = array('li' => true);
-        
-        // we need TABLE and heading mismatch code
-        // we may need to make this more flexible for heading mismatch,
-        // or we can just create another info
-        
-        //////////////////////////////////////////////////////////////////////
-        // info[]->attr_transform_* : attribute transformations in elements
-        // pre is applied before any validation is done, post is done after
-        
-        $this->info['h1']->attr_transform_pre[] =
-        $this->info['h2']->attr_transform_pre[] =
-        $this->info['h3']->attr_transform_pre[] =
-        $this->info['h4']->attr_transform_pre[] =
-        $this->info['h5']->attr_transform_pre[] =
-        $this->info['h6']->attr_transform_pre[] =
-        $this->info['p'] ->attr_transform_pre[] = 
-                    new HTMLPurifier_AttrTransform_TextAlign();
-        
-        $this->info['bdo']->attr_transform_post[] =
-                    new HTMLPurifier_AttrTransform_BdoDir();
-        
-        $this->info['img']->attr_transform_post[] =
-                    new HTMLPurifier_AttrTransform_ImgRequired();
-        
-        //////////////////////////////////////////////////////////////////////
-        // info_attr_transform_* : global attribute transformation that is
-        // unconditionally called. Good for transformations that have complex
-        // start conditions
-        // pre is applied before any validation is done, post is done after
-        
-        $this->info_attr_transform_post[] = new HTMLPurifier_AttrTransform_Lang();
-        
-        // protect against stdclasses floating around
-        foreach ($this->info as $key => $obj) {
-            if (is_a($obj, 'stdclass')) {
-                unset($this->info[$key]);
-            }
-        }
-        
-        //////////////////////////////////////////////////////////////////////
-        // info_block_wrapper : wraps inline elements in block context
-        
-        $block_wrapper = $config->get('HTML', 'BlockWrapper');
-        if (isset($e_Block->elements[$block_wrapper])) {
-            $this->info_block_wrapper = $block_wrapper;
-        } else {
-            trigger_error('Cannot use non-block element as block wrapper.',
-                E_USER_ERROR);
-        }
-        
-        //////////////////////////////////////////////////////////////////////
-        // info_parent : parent element of the HTML fragment
-        
-        $parent = $config->get('HTML', 'Parent');
-        if (isset($this->info[$parent])) {
-            $this->info_parent = $parent;
-        } else {
-            trigger_error('Cannot use unrecognized element as parent.',
-                E_USER_ERROR);
-        }
-        $this->info_parent_def = $this->info[$this->info_parent];
-        
-        //////////////////////////////////////////////////////////////////////
-        // %HTML.Allowed(Elements|Attributes) : cut non-allowed elements
-        
-        $allowed_elements = $config->get('HTML', 'AllowedElements');
-        if (is_array($allowed_elements)) {
-            foreach ($this->info as $name => $d) {
-                if(!isset($allowed_elements[$name])) unset($this->info[$name]);
-            }
-        }
-        $allowed_attributes = $config->get('HTML', 'AllowedAttributes');
-        if (is_array($allowed_attributes)) {
-            foreach ($this->info_global_attr as $attr_key => $info) {
-                if (!isset($allowed_attributes["*.$attr_key"])) {
-                    unset($this->info_global_attr[$attr_key]);
-                }
-            }
-            foreach ($this->info as $tag => $info) {
-                foreach ($info->attr as $attr => $attr_info) {
-                    if (!isset($allowed_attributes["$tag.$attr"])) {
-                        unset($this->info[$tag]->attr[$attr]);
-                    }
-                }
-            }
-        }
-    }
-    
-    function setAttrForTableElements($attr, $def) {
-        $this->info['col']->attr[$attr] = 
-        $this->info['colgroup']->attr[$attr] = 
-        $this->info['tbody']->attr[$attr] = 
-        $this->info['td']->attr[$attr] = 
-        $this->info['tfoot']->attr[$attr] = 
-        $this->info['th']->attr[$attr] = 
-        $this->info['thead']->attr[$attr] = 
-        $this->info['tr']->attr[$attr] = $def;
-    }
-    
-}
-
-/**
- * Structure that stores an element definition.
- */
-class HTMLPurifier_ElementDef
-{
-    
-    /**
-     * Associative array of attribute name to HTMLPurifier_AttrDef
-     * @public
-     */
-    var $attr = array();
-    
-    /**
-     * List of tag's HTMLPurifier_AttrTransform to be done before validation
-     * @public
-     */
-    var $attr_transform_pre = array();
-    
-    /**
-     * List of tag's HTMLPurifier_AttrTransform to be done after validation
-     * @public
-     */
-    var $attr_transform_post = array();
-    
-    /**
-     * Lookup table of tags that close this tag.
-     * @public
-     */
-    var $auto_close = array();
-    
-    /**
-     * HTMLPurifier_ChildDef of this tag.
-     * @public
-     */
-    var $child;
-    
-    /**
-     * Type of the tag: inline or block or unknown?
-     * @public
-     */
-    var $type = 'unknown';
-    
-    /**
-     * Lookup table of tags excluded from all descendants of this tag.
-     * @public
-     */
-    var $excludes = array();
-    
-}
-
-?>
+<?php
+
+// components
+require_once 'HTMLPurifier/HTMLModuleManager.php';
+
+// this definition and its modules MUST NOT define configuration directives
+// outside of the HTML or Attr namespaces
+
+// will be superceded by more accurate doctype declaration schemes
+HTMLPurifier_ConfigSchema::define(
+    'HTML', 'Strict', false, 'bool',
+    'Determines whether or not to use Transitional (loose) or Strict rulesets. '.
+    'This directive has been available since 1.3.0.'
+);
+
+HTMLPurifier_ConfigSchema::define(
+    'HTML', 'BlockWrapper', 'p', 'string',
+    'String name of element to wrap inline elements that are inside a block '.
+    'context.  This only occurs in the children of blockquote in strict mode. '.
+    'Example: by default value, <code>&lt;blockquote&gt;Foo&lt;/blockquote&gt;</code> '.
+    'would become <code>&lt;blockquote&gt;&lt;p&gt;Foo&lt;/p&gt;&lt;/blockquote&gt;</code>. The '.
+    '<code>&lt;p&gt;</code> tags can be replaced '.
+    'with whatever you desire, as long as it is a block level element. '.
+    'This directive has been available since 1.3.0.'
+);
+
+HTMLPurifier_ConfigSchema::define(
+    'HTML', 'Parent', 'div', 'string',
+    'String name of element that HTML fragment passed to library will be '.
+    'inserted in.  An interesting variation would be using span as the '.
+    'parent element, meaning that only inline tags would be allowed. '.
+    'This directive has been available since 1.3.0.'
+);
+
+HTMLPurifier_ConfigSchema::define(
+    'HTML', 'AllowedElements', null, 'lookup/null',
+    'If HTML Purifier\'s tag set is unsatisfactory for your needs, you '.
+    'can overload it with your own list of tags to allow.  Note that this '.
+    'method is subtractive: it does its job by taking away from HTML Purifier '.
+    'usual feature set, so you cannot add a tag that HTML Purifier never '.
+    'supported in the first place (like embed, form or head).  If you change this, you '.
+    'probably also want to change %HTML.AllowedAttributes. '.
+    '<strong>Warning:</strong> If another directive conflicts with the '.
+    'elements here, <em>that</em> directive will win and override. '.
+    'This directive has been available since 1.3.0.'
+);
+
+HTMLPurifier_ConfigSchema::define(
+    'HTML', 'AllowedAttributes', null, 'lookup/null',
+    'IF HTML Purifier\'s attribute set is unsatisfactory, overload it! '.
+    'The syntax is \'tag.attr\' or \'*.attr\' for the global attributes '.
+    '(style, id, class, dir, lang, xml:lang).'.
+    '<strong>Warning:</strong> If another directive conflicts with the '.
+    'elements here, <em>that</em> directive will win and override. For '.
+    'example, %HTML.EnableAttrID will take precedence over *.id in this '.
+    'directive.  You must set that directive to true before you can use '.
+    'IDs at all. This directive has been available since 1.3.0.'
+);
+
+/**
+ * Definition of the purified HTML that describes allowed children,
+ * attributes, and many other things.
+ * 
+ * Conventions:
+ * 
+ * All member variables that are prefixed with info
+ * (including the main $info array) are used by HTML Purifier internals
+ * and should not be directly edited when customizing the HTMLDefinition.
+ * They can usually be set via configuration directives or custom
+ * modules.
+ * 
+ * On the other hand, member variables without the info prefix are used
+ * internally by the HTMLDefinition and MUST NOT be used by other HTML
+ * Purifier internals. Many of them, however, are public, and may be
+ * edited by userspace code to tweak the behavior of HTMLDefinition.
+ * 
+ * HTMLPurifier_Printer_HTMLDefinition is a notable exception to this
+ * rule: in the interest of comprehensiveness, it will sniff everything.
+ */
+class HTMLPurifier_HTMLDefinition
+{
+    
+    /** FULLY-PUBLIC VARIABLES */
+    
+    /**
+     * Associative array of element names to HTMLPurifier_ElementDef
+     * @public
+     */
+    var $info = array();
+    
+    /**
+     * Associative array of global attribute name to attribute definition.
+     * @public
+     */
+    var $info_global_attr = array();
+    
+    /**
+     * String name of parent element HTML will be going into.
+     * @public
+     */
+    var $info_parent = 'div';
+    
+    /**
+     * Definition for parent element, allows parent element to be a
+     * tag that's not allowed inside the HTML fragment.
+     * @public
+     */
+    var $info_parent_def;
+    
+    /**
+     * String name of element used to wrap inline elements in block context
+     * @note This is rarely used except for BLOCKQUOTEs in strict mode
+     * @public
+     */
+    var $info_block_wrapper = 'p';
+    
+    /**
+     * Associative array of deprecated tag name to HTMLPurifier_TagTransform
+     * @public
+     */
+    var $info_tag_transform = array();
+    
+    /**
+     * Indexed list of HTMLPurifier_AttrTransform to be performed before validation.
+     * @public
+     */
+    var $info_attr_transform_pre = array();
+    
+    /**
+     * Indexed list of HTMLPurifier_AttrTransform to be performed after validation.
+     * @public
+     */
+    var $info_attr_transform_post = array();
+    
+    /**
+     * Nested lookup array of content set name (Block, Inline) to
+     * element name to whether or not it belongs in that content set.
+     * @public
+     */
+    var $info_content_sets = array();
+    
+    
+    
+    /** PUBLIC BUT INTERNAL VARIABLES */
+    
+    var $setup = false; /**< Has setup() been called yet? */
+    var $config; /**< Temporary instance of HTMLPurifier_Config */
+    
+    var $manager; /**< Instance of HTMLPurifier_HTMLModuleManager */
+    
+    /**
+     * Performs low-cost, preliminary initialization.
+     * @param $config Instance of HTMLPurifier_Config
+     */
+    function HTMLPurifier_HTMLDefinition(&$config) {
+        $this->config =& $config;
+        $this->manager = new HTMLPurifier_HTMLModuleManager();
+    }
+    
+    /**
+     * Processes internals into form usable by HTMLPurifier internals. 
+     * Modifying the definition after calling this function should not
+     * be done.
+     */
+    function setup() {
+        
+        // multiple call guard
+        if ($this->setup) {return;} else {$this->setup = true;}
+        
+        $this->processModules();
+        $this->setupConfigStuff();
+        
+        unset($this->config);
+        unset($this->manager);
+        
+    }
+    
+    /**
+     * Extract out the information from the manager
+     */
+    function processModules() {
+        
+        $this->manager->setup($this->config);
+        
+        foreach ($this->manager->activeModules as $module) {
+            foreach($module->info_tag_transform         as $k => $v) $this->info_tag_transform[$k]      = $v;
+            foreach($module->info_attr_transform_pre    as $k => $v) $this->info_attr_transform_pre[$k] = $v;
+            foreach($module->info_attr_transform_post   as $k => $v) $this->info_attr_transform_post[$k]= $v;
+        }
+        
+        $this->info = $this->manager->getElements($this->config);
+        $this->info_content_sets = $this->manager->contentSets->lookup;
+        
+    }
+    
+    /**
+     * Sets up stuff based on config. We need a better way of doing this.
+     */
+    function setupConfigStuff() {
+        
+        $block_wrapper = $this->config->get('HTML', 'BlockWrapper');
+        if (isset($this->info_content_sets['Block'][$block_wrapper])) {
+            $this->info_block_wrapper = $block_wrapper;
+        } else {
+            trigger_error('Cannot use non-block element as block wrapper.',
+                E_USER_ERROR);
+        }
+        
+        $parent = $this->config->get('HTML', 'Parent');
+        $def = $this->manager->getElement($parent, $this->config);
+        if ($def) {
+            $this->info_parent = $parent;
+            $this->info_parent_def = $def;
+        } else {
+            trigger_error('Cannot use unrecognized element as parent.',
+                E_USER_ERROR);
+            $this->info_parent_def = $this->manager->getElement(
+                $this->info_parent, $this->config);
+        }
+        
+        // support template text
+        $support = "(for information on implementing this, see the ".
+                   "support forums) ";
+        
+        // setup allowed elements, SubtractiveWhitelist module
+        $allowed_elements = $this->config->get('HTML', 'AllowedElements');
+        if (is_array($allowed_elements)) {
+            foreach ($this->info as $name => $d) {
+                if(!isset($allowed_elements[$name])) unset($this->info[$name]);
+                unset($allowed_elements[$name]);
+            }
+            // emit errors
+            foreach ($allowed_elements as $element => $d) {
+                trigger_error("Element '$element' is not supported $support", E_USER_WARNING);
+            }
+        }
+        
+        $allowed_attributes = $this->config->get('HTML', 'AllowedAttributes');
+        $allowed_attributes_mutable = $allowed_attributes; // by copy!
+        if (is_array($allowed_attributes)) {
+            foreach ($this->info_global_attr as $attr_key => $info) {
+                if (!isset($allowed_attributes["*.$attr_key"])) {
+                    unset($this->info_global_attr[$attr_key]);
+                } elseif (isset($allowed_attributes_mutable["*.$attr_key"])) {
+                    unset($allowed_attributes_mutable["*.$attr_key"]);
+                }
+            }
+            foreach ($this->info as $tag => $info) {
+                foreach ($info->attr as $attr => $attr_info) {
+                    if (!isset($allowed_attributes["$tag.$attr"]) &&
+                        !isset($allowed_attributes["*.$attr"])) {
+                        unset($this->info[$tag]->attr[$attr]);
+                    } else {
+                        if (isset($allowed_attributes_mutable["$tag.$attr"])) {
+                            unset($allowed_attributes_mutable["$tag.$attr"]);
+                        } elseif (isset($allowed_attributes_mutable["*.$attr"])) {
+                            unset($allowed_attributes_mutable["*.$attr"]);
+                        }
+                    }
+                }
+            }
+            // emit errors
+            foreach ($allowed_attributes_mutable as $elattr => $d) {
+                list($element, $attribute) = explode('.', $elattr);
+                if ($element == '*') {
+                    trigger_error("Global attribute '$attribute' is not ".
+                        "supported in any elements $support",
+                        E_USER_WARNING);
+                } else {
+                    trigger_error("Attribute '$attribute' in element '$element' not supported $support",
+                        E_USER_WARNING);
+                }
+            }
+        }
+        
+    }
+    
+    
+}
+
+?>
--- a/library/HTMLPurifier/HTMLModule.php
+++ b/library/HTMLPurifier/HTMLModule.php
@@ -0,0 +1,125 @@
+<?php
+
+/**
+ * Represents an XHTML 1.1 module, with information on elements, tags
+ * and attributes.
+ * @note Even though this is technically XHTML 1.1, it is also used for
+ *       regular HTML parsing. We are using modulization as a convenient
+ *       way to represent the internals of HTMLDefinition, and our
+ *       implementation is by no means conforming and does not directly
+ *       use the normative DTDs or XML schemas.
+ * @note The public variables in a module should almost directly
+ *       correspond to the variables in HTMLPurifier_HTMLDefinition.
+ *       However, the prefix info carries no special meaning in these
+ *       objects (include it anyway if that's the correspondence though).
+ */
+
+class HTMLPurifier_HTMLModule
+{
+    /**
+     * Short unique string identifier of the module
+     */
+    var $name;
+    
+    /**
+     * Dynamically set integer that specifies when the module was loaded in.
+     */
+    var $order;
+    
+    /**
+     * Informally, a list of elements this module changes. Not used in
+     * any significant way.
+     * @protected
+     */
+    var $elements = array();
+    
+    /**
+     * Associative array of element names to element definitions.
+     * Some definitions may be incomplete, to be merged in later
+     * with the full definition.
+     * @public
+     */
+    var $info = array();
+    
+    /**
+     * Associative array of content set names to content set additions.
+     * This is commonly used to, say, add an A element to the Inline
+     * content set. This corresponds to an internal variable $content_sets
+     * and NOT info_content_sets member variable of HTMLDefinition.
+     * @public
+     */
+    var $content_sets = array();
+    
+    /**
+     * Associative array of attribute collection names to attribute
+     * collection additions. More rarely used for adding attributes to
+     * the global collections. Example is the StyleAttribute module adding
+     * the style attribute to the Core. Corresponds to HTMLDefinition's
+     * attr_collections->info, since the object's data is only info,
+     * with extra behavior associated with it.
+     * @public
+     */
+    var $attr_collections = array();
+    
+    /**
+     * Associative array of deprecated tag name to HTMLPurifier_TagTransform
+     * @public
+     */
+    var $info_tag_transform = array();
+    
+    /**
+     * List of HTMLPurifier_AttrTransform to be performed before validation.
+     * @public
+     */
+    var $info_attr_transform_pre = array();
+    
+    /**
+     * List of HTMLPurifier_AttrTransform to be performed after validation.
+     * @public
+     */
+    var $info_attr_transform_post = array();
+    
+    /**
+     * Boolean flag that indicates whether or not getChildDef is implemented.
+     * For optimization reasons: may save a call to a function. Be sure
+     * to set it if you do implement getChildDef(), otherwise it will have
+     * no effect!
+     * @public
+     */
+    var $defines_child_def = false;
+    
+    /**
+     * Retrieves a proper HTMLPurifier_ChildDef subclass based on 
+     * content_model and content_model_type member variables of
+     * the HTMLPurifier_ElementDef class. There is a similar function
+     * in HTMLPurifier_HTMLDefinition.
+     * @param $def HTMLPurifier_ElementDef instance
+     * @return HTMLPurifier_ChildDef subclass
+     * @public
+     */
+    function getChildDef($def) {return false;}
+    
+    /**
+     * Hook method that lets module perform arbitrary operations on
+     * HTMLPurifier_HTMLDefinition before the module gets processed.
+     * @param $definition Reference to HTMLDefinition being setup
+     */
+    function preProcess(&$definition) {}
+    
+    /**
+     * Hook method that lets module perform arbitrary operations
+     * on HTMLPurifier_HTMLDefinition after the module gets processed.
+     * @param $definition Reference to HTMLDefinition being setup
+     */
+    function postProcess(&$definition) {}
+    
+    /**
+     * Hook method that is called when a module gets registered to
+     * the definition.
+     * @param $definition Reference to HTMLDefinition being setup
+     */
+    function setup(&$definition) {}
+    
+}
+
+?>
--- a/library/HTMLPurifier/HTMLModule/Bdo.php
+++ b/library/HTMLPurifier/HTMLModule/Bdo.php
@@ -0,0 +1,43 @@
+<?php
+
+require_once 'HTMLPurifier/HTMLModule.php';
+require_once 'HTMLPurifier/AttrTransform/BdoDir.php';
+
+/**
+ * XHTML 1.1 Bi-directional Text Module, defines elements that
+ * declare directionality of content. Text Extension Module.
+ */
+class HTMLPurifier_HTMLModule_Bdo extends HTMLPurifier_HTMLModule
+{
+    
+    var $name = 'Bdo';
+    var $elements = array('bdo');
+    var $info = array();
+    var $content_sets = array('Inline' => 'bdo');
+    var $attr_collections = array(
+        'I18N' => array('dir' => false)
+    );
+    
+    function HTMLPurifier_HTMLModule_Bdo() {
+        $dir = new HTMLPurifier_AttrDef_Enum(array('ltr','rtl'), false);
+        $this->attr_collections['I18N']['dir'] = $dir;
+        $this->info['bdo'] = new HTMLPurifier_ElementDef();
+        $this->info['bdo']->attr = array(
+            0 => array('Core', 'Lang'),
+            'dir' => $dir, // required
+            // The Abstract Module specification has the attribute
+            // inclusions wrong for bdo: bdo allows
+            // xml:lang too (and we'll toss in lang for good measure,
+            // though it is not allowed for XHTML 1.1, this will
+            // be managed with a global attribute transform)
+        );
+        $this->info['bdo']->content_model = '#PCDATA | Inline';
+        $this->info['bdo']->content_model_type = 'optional';
+        // provides fallback behavior if dir's missing (dir is required)
+        $this->info['bdo']->attr_transform_post['required-dir'] =
+            new HTMLPurifier_AttrTransform_BdoDir();
+    }
+    
+}
+
+?>
--- a/library/HTMLPurifier/HTMLModule/CommonAttributes.php
+++ b/library/HTMLPurifier/HTMLModule/CommonAttributes.php
@@ -0,0 +1,31 @@
+<?php
+
+class HTMLPurifier_HTMLModule_CommonAttributes extends HTMLPurifier_HTMLModule
+{
+    var $name = 'CommonAttributes';
+    
+    var $attr_collections = array(
+        'Core' => array(
+            0 => array('Style'),
+            // 'xml:space' => false,
+            'class' => 'NMTOKENS',
+            'id' => 'ID',
+            'title' => 'CDATA',
+        ),
+        'Lang' => array(
+            'xml:lang' => false, // see constructor
+        ),
+        'I18N' => array(
+            0 => array('Lang'), // proprietary, for xml:lang/lang
+        ),
+        'Common' => array(
+            0 => array('Core', 'I18N')
+        )
+    );
+    
+    function HTMLPurifier_HTMLModule_CommonAttributes() {
+        $this->attr_collections['Lang']['xml:lang'] = new HTMLPurifier_AttrDef_Lang();
+    }
+}
+
+?>
--- a/Show More
+++ b/Show More