2006-07-24 01:50:02 +00:00
|
|
|
<?php
|
|
|
|
|
2006-07-30 19:11:18 +00:00
|
|
|
/**
|
|
|
|
* Removes all unrecognized tags from the list of tokens.
|
2008-12-06 02:28:20 -05:00
|
|
|
*
|
2006-07-30 19:11:18 +00:00
|
|
|
* This strategy iterates through all the tokens and removes unrecognized
|
2006-08-02 02:26:01 +00:00
|
|
|
* tokens. If a token is not recognized but a TagTransform is defined for
|
|
|
|
* that element, the element will be transformed accordingly.
|
2006-07-30 19:11:18 +00:00
|
|
|
*/
|
|
|
|
|
2006-07-24 01:50:02 +00:00
|
|
|
class HTMLPurifier_Strategy_RemoveForeignElements extends HTMLPurifier_Strategy
|
|
|
|
{
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2008-01-05 00:10:43 +00:00
|
|
|
public function execute($tokens, $config, $context) {
|
2006-08-31 20:33:07 +00:00
|
|
|
$definition = $config->getHTMLDefinition();
|
2008-05-26 04:05:48 +00:00
|
|
|
$generator = new HTMLPurifier_Generator($config, $context);
|
2006-07-24 01:50:02 +00:00
|
|
|
$result = array();
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2009-02-19 19:17:49 -05:00
|
|
|
$escape_invalid_tags = $config->get('Core.EscapeInvalidTags');
|
|
|
|
$remove_invalid_img = $config->get('Core.RemoveInvalidImg');
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2008-06-25 23:12:19 -04:00
|
|
|
// currently only used to determine if comments should be kept
|
2009-02-19 19:17:49 -05:00
|
|
|
$trusted = $config->get('HTML.Trusted');
|
2011-12-26 15:34:42 +08:00
|
|
|
$comment_lookup = $config->get('HTML.AllowedComments');
|
|
|
|
$comment_regexp = $config->get('HTML.AllowedCommentsRegexp');
|
|
|
|
$check_comments = $comment_lookup !== array() || $comment_regexp !== null;
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2009-02-19 19:17:49 -05:00
|
|
|
$remove_script_contents = $config->get('Core.RemoveScriptContents');
|
|
|
|
$hidden_elements = $config->get('Core.HiddenElements');
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-07-11 20:42:58 +00:00
|
|
|
// remove script contents compatibility
|
|
|
|
if ($remove_script_contents === true) {
|
|
|
|
$hidden_elements['script'] = true;
|
|
|
|
} elseif ($remove_script_contents === false && isset($hidden_elements['script'])) {
|
|
|
|
unset($hidden_elements['script']);
|
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-20 21:39:28 +00:00
|
|
|
$attr_validator = new HTMLPurifier_AttrValidator();
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-16 19:31:45 +00:00
|
|
|
// removes tokens until it reaches a closing tag with its value
|
|
|
|
$remove_until = false;
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-21 14:44:26 +00:00
|
|
|
// converts comments into text tokens when this is equal to a tag name
|
|
|
|
$textify_comments = false;
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-25 01:56:00 +00:00
|
|
|
$token = false;
|
|
|
|
$context->register('CurrentToken', $token);
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-26 02:49:21 +00:00
|
|
|
$e = false;
|
2009-02-19 19:17:49 -05:00
|
|
|
if ($config->get('Core.CollectErrors')) {
|
2007-06-26 02:49:21 +00:00
|
|
|
$e =& $context->get('ErrorCollector');
|
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2006-07-24 01:50:02 +00:00
|
|
|
foreach($tokens as $token) {
|
2007-06-16 19:31:45 +00:00
|
|
|
if ($remove_until) {
|
|
|
|
if (empty($token->is_tag) || $token->name !== $remove_until) {
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
}
|
2006-07-24 01:50:02 +00:00
|
|
|
if (!empty( $token->is_tag )) {
|
2006-07-30 19:11:18 +00:00
|
|
|
// DEFINITION CALL
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-19 22:10:39 +00:00
|
|
|
// before any processing, try to transform the element
|
|
|
|
if (
|
|
|
|
isset($definition->info_tag_transform[$token->name])
|
|
|
|
) {
|
2007-06-26 02:49:21 +00:00
|
|
|
$original_name = $token->name;
|
2007-06-19 22:10:39 +00:00
|
|
|
// there is a transformation for this tag
|
|
|
|
// DEFINITION CALL
|
|
|
|
$token = $definition->
|
|
|
|
info_tag_transform[$token->name]->
|
|
|
|
transform($token, $config, $context);
|
2007-06-26 02:49:21 +00:00
|
|
|
if ($e) $e->send(E_NOTICE, 'Strategy_RemoveForeignElements: Tag transform', $original_name);
|
2007-06-19 22:10:39 +00:00
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2006-08-31 20:33:07 +00:00
|
|
|
if (isset($definition->info[$token->name])) {
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-20 21:39:28 +00:00
|
|
|
// mostly everything's good, but
|
|
|
|
// we need to make sure required attributes are in order
|
|
|
|
if (
|
2008-01-19 20:23:01 +00:00
|
|
|
($token instanceof HTMLPurifier_Token_Start || $token instanceof HTMLPurifier_Token_Empty) &&
|
2007-06-20 21:39:28 +00:00
|
|
|
$definition->info[$token->name]->required_attr &&
|
|
|
|
($token->name != 'img' || $remove_invalid_img) // ensure config option still works
|
|
|
|
) {
|
2007-06-27 02:03:15 +00:00
|
|
|
$attr_validator->validateToken($token, $config, $context);
|
2007-06-20 21:39:28 +00:00
|
|
|
$ok = true;
|
|
|
|
foreach ($definition->info[$token->name]->required_attr as $name) {
|
|
|
|
if (!isset($token->attr[$name])) {
|
|
|
|
$ok = false;
|
|
|
|
break;
|
|
|
|
}
|
2006-11-23 23:59:20 +00:00
|
|
|
}
|
2007-06-26 02:49:21 +00:00
|
|
|
if (!$ok) {
|
2007-06-26 19:33:37 +00:00
|
|
|
if ($e) $e->send(E_ERROR, 'Strategy_RemoveForeignElements: Missing required attribute', $name);
|
2007-06-26 02:49:21 +00:00
|
|
|
continue;
|
|
|
|
}
|
2007-06-20 21:39:28 +00:00
|
|
|
$token->armor['ValidateAttributes'] = true;
|
2006-11-23 23:59:20 +00:00
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2008-01-19 20:23:01 +00:00
|
|
|
if (isset($hidden_elements[$token->name]) && $token instanceof HTMLPurifier_Token_Start) {
|
2007-06-21 14:44:26 +00:00
|
|
|
$textify_comments = $token->name;
|
2008-01-19 20:23:01 +00:00
|
|
|
} elseif ($token->name === $textify_comments && $token instanceof HTMLPurifier_Token_End) {
|
2007-06-21 14:44:26 +00:00
|
|
|
$textify_comments = false;
|
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2006-08-15 23:58:18 +00:00
|
|
|
} elseif ($escape_invalid_tags) {
|
2007-06-26 02:49:21 +00:00
|
|
|
// invalid tag, generate HTML representation and insert in
|
2007-06-26 15:07:07 +00:00
|
|
|
if ($e) $e->send(E_WARNING, 'Strategy_RemoveForeignElements: Foreign element to text');
|
2006-07-24 01:50:02 +00:00
|
|
|
$token = new HTMLPurifier_Token_Text(
|
2008-05-26 04:05:48 +00:00
|
|
|
$generator->generateFromToken($token)
|
2006-07-24 01:50:02 +00:00
|
|
|
);
|
2006-08-15 23:58:18 +00:00
|
|
|
} else {
|
2007-06-16 19:31:45 +00:00
|
|
|
// check if we need to destroy all of the tag's children
|
|
|
|
// CAN BE GENERICIZED
|
2007-07-11 20:42:58 +00:00
|
|
|
if (isset($hidden_elements[$token->name])) {
|
2008-01-19 20:23:01 +00:00
|
|
|
if ($token instanceof HTMLPurifier_Token_Start) {
|
2007-06-16 19:31:45 +00:00
|
|
|
$remove_until = $token->name;
|
2008-01-19 20:23:01 +00:00
|
|
|
} elseif ($token instanceof HTMLPurifier_Token_Empty) {
|
2007-06-16 19:31:45 +00:00
|
|
|
// do nothing: we're still looking
|
|
|
|
} else {
|
|
|
|
$remove_until = false;
|
|
|
|
}
|
2007-07-11 20:42:58 +00:00
|
|
|
if ($e) $e->send(E_ERROR, 'Strategy_RemoveForeignElements: Foreign meta element removed');
|
2007-06-26 02:49:21 +00:00
|
|
|
} else {
|
2007-06-26 15:07:07 +00:00
|
|
|
if ($e) $e->send(E_ERROR, 'Strategy_RemoveForeignElements: Foreign element removed');
|
2007-06-16 19:31:45 +00:00
|
|
|
}
|
2006-08-15 23:58:18 +00:00
|
|
|
continue;
|
2006-07-24 01:50:02 +00:00
|
|
|
}
|
2008-01-19 20:23:01 +00:00
|
|
|
} elseif ($token instanceof HTMLPurifier_Token_Comment) {
|
2007-06-21 14:44:26 +00:00
|
|
|
// textify comments in script tags when they are allowed
|
|
|
|
if ($textify_comments !== false) {
|
|
|
|
$data = $token->data;
|
|
|
|
$token = new HTMLPurifier_Token_Text($data);
|
2011-12-26 15:34:42 +08:00
|
|
|
} elseif ($trusted || $check_comments) {
|
|
|
|
// always cleanup comments
|
|
|
|
$trailing_hyphen = false;
|
2008-06-25 23:12:19 -04:00
|
|
|
if ($e) {
|
|
|
|
// perform check whether or not there's a trailing hyphen
|
|
|
|
if (substr($token->data, -1) == '-') {
|
2011-12-26 15:34:42 +08:00
|
|
|
$trailing_hyphen = true;
|
2008-06-25 23:12:19 -04:00
|
|
|
}
|
|
|
|
}
|
|
|
|
$token->data = rtrim($token->data, '-');
|
|
|
|
$found_double_hyphen = false;
|
|
|
|
while (strpos($token->data, '--') !== false) {
|
2011-12-26 15:34:42 +08:00
|
|
|
$found_double_hyphen = true;
|
2008-06-25 23:12:19 -04:00
|
|
|
$token->data = str_replace('--', '-', $token->data);
|
|
|
|
}
|
2011-12-26 15:34:42 +08:00
|
|
|
if ($trusted || !empty($comment_lookup[trim($token->data)]) || ($comment_regexp !== NULL && preg_match($comment_regexp, trim($token->data)))) {
|
|
|
|
// OK good
|
|
|
|
if ($e) {
|
|
|
|
if ($trailing_hyphen) {
|
|
|
|
$e->send(E_NOTICE, 'Strategy_RemoveForeignElements: Trailing hyphen in comment removed');
|
|
|
|
}
|
|
|
|
if ($found_double_hyphen) {
|
|
|
|
$e->send(E_NOTICE, 'Strategy_RemoveForeignElements: Hyphens in comment collapsed');
|
|
|
|
}
|
|
|
|
}
|
|
|
|
} else {
|
|
|
|
if ($e) {
|
|
|
|
$e->send(E_NOTICE, 'Strategy_RemoveForeignElements: Comment removed');
|
|
|
|
}
|
|
|
|
continue;
|
|
|
|
}
|
2007-06-21 14:44:26 +00:00
|
|
|
} else {
|
|
|
|
// strip comments
|
2007-06-26 19:33:37 +00:00
|
|
|
if ($e) $e->send(E_NOTICE, 'Strategy_RemoveForeignElements: Comment removed');
|
2007-06-21 14:44:26 +00:00
|
|
|
continue;
|
|
|
|
}
|
2008-01-19 20:23:01 +00:00
|
|
|
} elseif ($token instanceof HTMLPurifier_Token_Text) {
|
2006-07-24 01:50:02 +00:00
|
|
|
} else {
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
$result[] = $token;
|
|
|
|
}
|
2007-06-26 02:49:21 +00:00
|
|
|
if ($remove_until && $e) {
|
|
|
|
// we removed tokens until the end, throw error
|
|
|
|
$e->send(E_ERROR, 'Strategy_RemoveForeignElements: Token removed to end', $remove_until);
|
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2007-06-25 01:56:00 +00:00
|
|
|
$context->destroy('CurrentToken');
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2006-07-24 01:50:02 +00:00
|
|
|
return $result;
|
|
|
|
}
|
2008-12-06 02:28:20 -05:00
|
|
|
|
2006-07-24 01:50:02 +00:00
|
|
|
}
|
|
|
|
|
2008-12-06 04:24:59 -05:00
|
|
|
// vim: et sw=4 sts=4
|