




* Title: HTML 相关的正则表达式工具类 *
** Description: 包括过滤 HTML 标记,转换 HTML 标记,替换特定 HTML 标记 *
** Copyright: Copyright (c) 2006 *
* * @author hejian * @version 1.0 * @createtime 2006-10-16 */ public class HtmlRegexpUtil { private final static String regxpForHtml = "<([^>]*)>"; // 过滤所有以<开头以>结尾的标签 private final static String regxpForImgTag = "<\\s*img\\s+([^>]*)\\s*>"; // 找出 IMG 标签 private final static String regxpForImaTagSrcAttrib = "src=\"([^\"]+)\""; // 找出 IMG 标签的 SRC 属性 /** * */ public HtmlRegexpUtil() { // TODO Auto-generated constructor stub } /** * * 基本功能:替换标记以正常显示 ** * @param input * @return String */ public String replaceTag(String input) { if (!hasSpecialChars(input)) {
* * @param input * @return boolean */ public boolean hasSpecialChars(String input) { boolean flag = false; if ((input != null) && (input.length() > 0)) { char c; for (int i = 0; i <= input.length() - 1; i++) { c = input.charAt(i); switch (c) { case '>': flag = true;
* * @param str * @return String */ public static String filterHtml(String str) { Pattern pattern = Pattern.compile(regxpForHtml); Matcher matcher = pattern.matcher(str); StringBuffer sb = new StringBuffer(); boolean result1 = matcher.find(); while (result1) { matcher.appendReplacement(sb, ""); result1 = matcher.find(); } matcher.appendTail(sb); return sb.toString(); } /** * * 基本功能:过滤指定标签 *
* * @param str * @param tag
* * @param str * @param beforeTag * * @param tagAttrib * * @param startTag * * @param endTag * * @return String * @如:替换 img 标签的 src 属性值为[img]属性值[/img] */ public static String replaceHtmlTag(String str, String beforeTag, 要替换的标签属性值 新标签开始标记 新标签结束标记 String tagAttrib, String startTag, String endTag) { String regxpForTag = "<\\s*" + beforeTag + "\\s+([^>]*)\\s*>"; String regxpForTagAttrib = tagAttrib + "=\"([^\"]+)\""; Pattern patternForTag = Pattern.compile(regxpForTag); Pattern patternForAttrib = Pattern.compile(regxpForTagAttrib); Matcher matcherForTag = patternForTag.matcher(str); StringBuffer sb = new StringBuffer(); boolean result = matcherForTag.find(); while (result) {