July 2002
Intermediate to advanced
608 pages
15h 46m
English
Credit: Itamar Shtull-Trauring
You have received some HTML input from a user and need to make sure that the HTML is clean. You want to allow only safe tags, to ensure that tags needing closure are indeed closed, and, ideally, to strip out any Javascript that might be part of the page.
The sgmllib module helps with cleaning up the HTML
tags, but we still have to fight against the Javascript:
import sgmllib, string class StrippingParser(sgmllib.SGMLParser): # These are the HTML tags that we will leave intact valid_tags = ('b', 'a', 'i', 'br', 'p') tolerate_missing_closing_tags = ('br', 'p') from htmlentitydefs import entitydefs # replace entitydefs from sgmllib def _ _init_ _(self): sgmllib.SGMLParser._ _init_ _(self) self.result = [] self.endTagList = [] def handle_data(self, data): self.result.append(data) def handle_charref(self, name): self.result.append("&#%s;" % name) def handle_entityref(self, name): x = ';' * self.entitydefs.has_key(name) self.result.append("&%s%s" % (name, x)) def unknown_starttag(self, tag, attrs): """ Delete all tags except for legal ones. """ if tag in self.valid_tags: self.result.append('<' + tag) for k, v in attrs: if string.lower(k[0:2]) != 'on' and string.lower( v[0:10]) != 'javascript': self.result.append(' %s="%s"' % (k, v)) self.result.append('>') if tag not in self.tolerate_missing_closing_tags: endTag = '</%s>' % tag self.endTagList.insert(0,endTag) def unknown_endtag(self, tag): ...Read now
Unlock full access