solr是基于 lucence开发的应用,如果query中带有非法字符串,结果很可能是检索出所有内容或者直接报错,所以你对用户的输入必须要先做处理。输入星号,能够检索出所有内容;输入加号,则会报错。
官方的处理办法(java,因为solr是java开发的):
-
https://svn.apache.org/repos/asf/lucene/dev/trunk/solr/solrj/src/java/org/apache/solr/client/solrj/util/ClientUtils.java
-
-
public static String escapeQueryChars(String s) {
-
StringBuilder sb = new StringBuilder();
-
for (int i = 0; i < s.length(); i++) {
-
char c = s.charAt(i);
-
// These characters are part of the query syntax and must be escaped
-
if (c == '\' || c == '+' || c == '-' || c == '!' || c == '(' || c == ')' || c == ':'
-
|| c == '^' || c == '[' || c == ']' || c == '"' || c == '{' || c == '}' || c == '~'
-
|| c == '*' || c == '?' || c == '|' || c == '&' || c == ';' || c == '/'
-
|| Character.isWhitespace(c)) {
-
sb.append('\');
-
}
-
sb.append(c);
-
}
-
return sb.toString();
-
}
翻译的php版本(利用preg_replace函数进行正则替换):
-
static public function escape($value)
-
{
-
//list taken from http://lucene.apache.org/java/docs/queryparsersyntax.html#Escaping%20Special%20Characters
-
$pattern = '/(+|-|&|||!|(|)|{|}|[|]|^|"|~|*|?|:|;|~|/)/';
-
$replace = '\$1';
-
-
return preg_replace($pattern, $replace, $value);
-
}
翻译后的python版本:
-
import re
-
def escape_solr(word):
-
return re.sub('(\|+|-|&||||!|(|)|{|}|[|]|^|"|~|*|?|:|;|/|~)','\1', word )
C#版
/// <summary> /// solr查询字符串特殊字符转义 /// </summary> /// <param name="s"></param> /// <returns></returns> public static String escapeQueryChars(String s) { StringBuilder sb = new StringBuilder(); var ss = s.ToCharArray(); foreach (var c in ss) { // These characters are part of the query syntax and must be escaped if (c == '\' || c == '+' || c == '-' || c == '!' || c == '(' || c == ')' || c == ':' || c == '^' || c == '[' || c == ']' || c == '"' || c == '{' || c == '}' || c == '~' || c == '*' || c == '?' || c == '|' || c == '&' || c == ';' || c == '/' ) { sb.Append('\'); } sb.Append(c); } return sb.ToString(); }