/*------------------------------------------------------------------------------ * Copyright (C) 2003-2006 Ben van Klinken and the CLucene Team * * Distributable under the terms of either the Apache License (Version 2.0) or * the GNU Lesser General Public License, as specified in the COPYING file. ------------------------------------------------------------------------------*/ #ifndef _lucene_queryParser_QueryParser_ #define _lucene_queryParser_QueryParser_ #include "CLucene/util/Array.h" #include "QueryParserTokenManager.h" #include "CLucene/document/DateTools.h" #include "CLucene/util/VoidMap.h" #include "CLucene/util/VoidList.h" CL_CLASS_DEF(index,Term) CL_CLASS_DEF(analysis,Analyzer) CL_CLASS_DEF(search,Query) CL_CLASS_DEF(search,BooleanClause) CL_NS_DEF(queryParser) class QueryParserConstants; /** * This class is generated by JavaCC. The most important method is * {@link #parse(String)}. * * The syntax for query strings is as follows: * A Query is a series of clauses. * A clause may be prefixed by: *
+) or a minus (-) sign, indicating
* that the clause is required or prohibited respectively; or
* +/- prefix to require any of a set of
* terms.
*
* Query ::= ( Clause )*
* Clause ::= ["+", "-"] [<TERM> ":"] ( <TERM> | "(" Query ")" )
*
*
* * Examples of appropriately formatted queries can be found in the query syntax * documentation. *
* ** In {@link RangeQuery}s, QueryParser tries to detect date values, e.g. * date:[6/1/2005 TO 6/4/2005] produces a range query that searches * for "date" fields between 2005-06-01 and 2005-06-04. Note that the format * of the accepted input depends on {@link #setLocale(Locale) the locale}. * By default a date is converted into a search term using the deprecated * {@link DateField} for compatibility reasons. * To use the new {@link DateTools} to convert dates, a * {@link org.apache.lucene.document.DateTools.Resolution} has to be set. *
** The date resolution that shall be used for RangeQueries can be set * using {@link #setDateResolution(DateTools.Resolution)} * or {@link #setDateResolution(String, DateTools.Resolution)}. The former * sets the default date resolution for all fields, whereas the latter can * be used to set field specific date resolutions. Field specific date * resolutions take, if set, precedence over the default date resolution. *
** If you use neither {@link DateField} nor {@link DateTools} in your * index, you can create your own * query parser that inherits QueryParser and overwrites * {@link #getRangeQuery(String, String, String, boolean)} to * use a different method for date conversion. *
* *Note that QueryParser is not thread-safe.
* * @author Brian Goetz * @author Peter Halacsy * @author Tatu Saloranta */ class CLUCENE_EXPORT QueryParser : public virtual QueryParserConstants { private: LUCENE_STATIC_CONSTANT(int32_t, CONJ_NONE=0); LUCENE_STATIC_CONSTANT(int32_t, CONJ_AND=1); LUCENE_STATIC_CONSTANT(int32_t, CONJ_OR=2); LUCENE_STATIC_CONSTANT(int32_t, MOD_NONE=0); LUCENE_STATIC_CONSTANT(int32_t, MOD_NOT=10); LUCENE_STATIC_CONSTANT(int32_t, MOD_REQ=11); public: /** The default operator for parsing queries. * Use {@link QueryParser#setDefaultOperator} to change it. */ enum Operator { OR_OPERATOR, AND_OPERATOR }; private: /** The actual operator that parser uses to combine query terms */ Operator _operator; bool lowercaseExpandedTerms; bool useOldRangeQuery; bool allowLeadingWildcard; bool enablePositionIncrements; CL_NS(analysis)::Analyzer* analyzer; TCHAR* field; int32_t phraseSlop; float_t fuzzyMinSim; int32_t fuzzyPrefixLength; //TODO: Locale locale = Locale.getDefault(); // the default date resolution CL_NS(document)::DateTools::Resolution dateResolution; // maps field names to date resolutions typedef CL_NS(util)::CLHashMaptrue to allow leading wildcard characters.
*
* When set, * or ? are allowed as
* the first character of a PrefixQuery and WildcardQuery.
* Note that this can produce very slow
* queries on big indexes.
*
* Default: false.
*/
void setAllowLeadingWildcard(const bool _allowLeadingWildcard);
/**
* @see #setAllowLeadingWildcard(boolean)
*/
bool getAllowLeadingWildcard() const;
/**
* Set to true to enable position increments in result query.
*
* When set, result phrase and multi-phrase queries will * be aware of position increments. * Useful when e.g. a StopFilter increases the position increment of * the token that follows an omitted token. *
* Default: false.
*/
void setEnablePositionIncrements(const bool _enable);
/**
* @see #setEnablePositionIncrements(boolean)
*/
bool getEnablePositionIncrements() const;
/**
* Sets the boolean operator of the QueryParser.
* In default mode (
* Depending on settings, prefix term may be lower-cased
* automatically. It will not go through the default Analyzer,
* however, since normal Analyzers are unlikely to work properly
* with wildcard templates.
*
* Can be overridden by extending classes, to provide custom handling for
* wildcard queries, which may be necessary due to missing analyzer calls.
*
* @param field Name of the field query will use.
* @param termStr Term token that contains one or more wild card
* characters (? or *), but is not simple prefix term
*
* @return Resulting {@link Query} built for the term
* @exception ParseException throw in overridden method to disallow
*/
virtual CL_NS(search)::Query* getWildcardQuery(const TCHAR* _field, TCHAR* termStr);
/**
* Factory method for generating a query (similar to
* {@link #getWildcardQuery}). Called when parser parses an input term
* token that uses prefix notation; that is, contains a single '*' wildcard
* character as its last character. Since this is a special case
* of generic wildcard term, and such a query can be optimized easily,
* this usually results in a different query object.
*
* Depending on settings, a prefix term may be lower-cased
* automatically. It will not go through the default Analyzer,
* however, since normal Analyzers are unlikely to work properly
* with wildcard templates.
*
* Can be overridden by extending classes, to provide custom handling for
* wild card queries, which may be necessary due to missing analyzer calls.
*
* @param field Name of the field query will use.
* @param termStr Term token to use for building term for the query
* (without trailing '*' character!)
*
* @return Resulting {@link Query} built for the term
* @exception ParseException throw in overridden method to disallow
*/
virtual CL_NS(search)::Query* getPrefixQuery(const TCHAR* _field, TCHAR* _termStr);
/**
* Factory method for generating a query (similar to
* {@link #getWildcardQuery}). Called when parser parses
* an input term token that has the fuzzy suffix (~) appended.
*
* @param field Name of the field query will use.
* @param termStr Term token to use for building term for the query
*
* @return Resulting {@link Query} built for the term
* @exception ParseException throw in overridden method to disallow
*/
virtual CL_NS(search)::Query* getFuzzyQuery(const TCHAR* _field, TCHAR* termStr, const float_t minSimilarity);
private:
/**
* Returns a String where the escape char has been
* removed, or kept only once if there was a double escape.
*
* Supports escaped unicode characters, e. g. translates
* OR_OPERATOR) terms without any modifiers
* are considered optional: for example capital of Hungary is equal to
* capital OR of OR Hungary.
* In AND_OPERATOR mode terms are considered to be in conjuction: the
* above mentioned query is parsed as capital AND of AND Hungary
*/
void setDefaultOperator(Operator _op);
/**
* Gets implicit operator setting, which will be either AND_OPERATOR
* or OR_OPERATOR.
*/
Operator getDefaultOperator() const;
/**
* Whether terms of wildcard, prefix, fuzzy and range queries are to be automatically
* lower-cased or not. Default is true.
*/
void setLowercaseExpandedTerms(const bool _lowercaseExpandedTerms);
/**
* @see #setLowercaseExpandedTerms(boolean)
*/
bool getLowercaseExpandedTerms() const;
/**
* By default QueryParser uses new ConstantScoreRangeQuery in preference to RangeQuery
* for range queries. This implementation is generally preferable because it
* a) Runs faster b) Does not have the scarcity of range terms unduly influence score
* c) avoids any "TooManyBooleanClauses" exception.
* However, if your application really needs to use the old-fashioned RangeQuery and the above
* points are not required then set this option to true
* Default is false.
*/
void setUseOldRangeQuery(const bool _useOldRangeQuery);
/**
* @see #setUseOldRangeQuery(boolean)
*/
bool getUseOldRangeQuery() const;
/**
* Set locale used by date range parsing.
*
void setLocale(const Locale _locale) {
locale = _locale;
}
* Returns current locale, allowing access by subclasses.
*
Locale getLocale() const {
return locale;
}
*/
/**
* Sets the default date resolution used by RangeQueries for fields for which no
* specific date resolutions has been set. Field specific resolutions can be set
* with {@link #setDateResolution(String, DateTools.Resolution)}.
*
* @param dateResolution the default date resolution to set
*/
void setDateResolution(const CL_NS(document)::DateTools::Resolution _dateResolution);
/**
* Sets the date resolution used by RangeQueries for a specific field.
*
* @param fieldName field for which the date resolution is to be set
* @param dateResolution date resolution to set
*/
void setDateResolution(const TCHAR* fieldName, const CL_NS(document)::DateTools::Resolution _dateResolution);
/**
* Returns the date resolution that is used by RangeQueries for the given field.
* Returns null (NO_RESOLUTION), if no default or field specific date resolution has been set
* for the given field.
*
*/
CL_NS(document)::DateTools::Resolution getDateResolution(const TCHAR* fieldName) const;
protected:
void addClause(std::vectorA to A.
*
* @memory caller is responsible to free the returned string
*
*/
TCHAR* discardEscapeChar(TCHAR* input, TCHAR* output=NULL);
/** Returns the numeric value of the hexadecimal character */
static int32_t hexToInt(TCHAR c);
struct JJCalls;
public:
/**
* Returns a String where those characters that QueryParser
* expects to be escaped are escaped by a preceding \.
*
* @memory caller is responsible to free the returned string
*/
static TCHAR* escape(const TCHAR* s);
// * Query ::= ( Clause )*
// * Clause ::= ["+", "-"] [