7 * This source file is subject to the new BSD license that is bundled
8 * with this package in the file LICENSE.txt.
9 * It is also available through the world-wide-web at this URL:
10 * http://framework.zend.com/license/new-bsd
11 * If you did not receive a copy of the license and are unable to
12 * obtain it through the world-wide-web, please send an email
13 * to license@zend.com so we can send you a copy immediately.
16 * @package Zend_Search_Lucene
17 * @subpackage Analysis
18 * @copyright Copyright (c) 2005-2009 Zend Technologies USA Inc. (http://www.zend.com)
19 * @license http://framework.zend.com/license/new-bsd New BSD License
20 * @version $Id: Utf8Num.php 16971 2009-07-22 18:05:45Z mikaelkael $
24 /** Zend_Search_Lucene_Analysis_Analyzer_Common */
25 require_once 'Zend/Search/Lucene/Analysis/Analyzer/Common.php';
30 * @package Zend_Search_Lucene
31 * @subpackage Analysis
32 * @copyright Copyright (c) 2005-2009 Zend Technologies USA Inc. (http://www.zend.com)
33 * @license http://framework.zend.com/license/new-bsd New BSD License
36 class Zend_Search_Lucene_Analysis_Analyzer_Common_Utf8Num extends Zend_Search_Lucene_Analysis_Analyzer_Common
39 * Current char position in an UTF-8 stream
46 * Current binary position in an UTF-8 stream
50 private $_bytePosition;
55 * @throws Zend_Search_Lucene_Exception
57 public function __construct()
59 if (@preg_match('/\pL/u', 'a') != 1) {
60 // PCRE unicode support is turned off
61 require_once 'Zend/Search/Lucene/Exception.php';
62 throw new Zend_Search_Lucene_Exception('Utf8Num analyzer needs PCRE unicode support to be enabled.');
69 public function reset()
72 $this->_bytePosition = 0;
74 // convert input into UTF-8
75 if (strcasecmp($this->_encoding, 'utf8' ) != 0 &&
76 strcasecmp($this->_encoding, 'utf-8') != 0 ) {
77 $this->_input = iconv($this->_encoding, 'UTF-8', $this->_input);
78 $this->_encoding = 'UTF-8';
83 * Tokenization stream API
85 * Returns null at the end of stream
87 * @return Zend_Search_Lucene_Analysis_Token|null
89 public function nextToken()
91 if ($this->_input === null) {
96 if (! preg_match('/[\p{L}\p{N}]+/u', $this->_input, $match, PREG_OFFSET_CAPTURE, $this->_bytePosition)) {
97 // It covers both cases a) there are no matches (preg_match(...) === 0)
98 // b) error occured (preg_match(...) === FALSE)
103 $matchedWord = $match[0][0];
105 // binary position of the matched word in the input stream
106 $binStartPos = $match[0][1];
108 // character position of the matched word in the input stream
109 $startPos = $this->_position +
110 iconv_strlen(substr($this->_input,
111 $this->_bytePosition,
112 $binStartPos - $this->_bytePosition),
114 // character postion of the end of matched word in the input stream
115 $endPos = $startPos + iconv_strlen($matchedWord, 'UTF-8');
117 $this->_bytePosition = $binStartPos + strlen($matchedWord);
118 $this->_position = $endPos;
120 $token = $this->normalize(new Zend_Search_Lucene_Analysis_Token($matchedWord, $startPos, $endPos));
121 } while ($token === null); // try again if token is skipped