mediatheque

This commit is contained in:
armansansd
2023-07-18 10:47:46 +01:00
commit 4b7c97a02e
3044 changed files with 544089 additions and 0 deletions
+25
View File
@@ -0,0 +1,25 @@
<?php
// autoload.php @generated by Composer
if (PHP_VERSION_ID < 50600) {
if (!headers_sent()) {
header('HTTP/1.1 500 Internal Server Error');
}
$err = 'Composer 2.3.0 dropped support for autoloading on PHP <5.6 and you are running '.PHP_VERSION.', please upgrade PHP or use Composer 2.2 LTS via "composer self-update --2.2". Aborting.'.PHP_EOL;
if (!ini_get('display_errors')) {
if (PHP_SAPI === 'cli' || PHP_SAPI === 'phpdbg') {
fwrite(STDERR, $err);
} elseif (!headers_sent()) {
echo $err;
}
}
trigger_error(
$err,
E_USER_ERROR
);
}
require_once __DIR__ . '/composer/autoload_real.php';
return ComposerAutoloaderInit6693564509f9a3fa6ed2c7bf76fdb017::getLoader();
+585
View File
@@ -0,0 +1,585 @@
<?php
/*
* This file is part of Composer.
*
* (c) Nils Adermann <naderman@naderman.de>
* Jordi Boggiano <j.boggiano@seld.be>
*
* For the full copyright and license information, please view the LICENSE
* file that was distributed with this source code.
*/
namespace Composer\Autoload;
/**
* ClassLoader implements a PSR-0, PSR-4 and classmap class loader.
*
* $loader = new \Composer\Autoload\ClassLoader();
*
* // register classes with namespaces
* $loader->add('Symfony\Component', __DIR__.'/component');
* $loader->add('Symfony', __DIR__.'/framework');
*
* // activate the autoloader
* $loader->register();
*
* // to enable searching the include path (eg. for PEAR packages)
* $loader->setUseIncludePath(true);
*
* In this example, if you try to use a class in the Symfony\Component
* namespace or one of its children (Symfony\Component\Console for instance),
* the autoloader will first look for the class under the component/
* directory, and it will then fallback to the framework/ directory if not
* found before giving up.
*
* This class is loosely based on the Symfony UniversalClassLoader.
*
* @author Fabien Potencier <fabien@symfony.com>
* @author Jordi Boggiano <j.boggiano@seld.be>
* @see https://www.php-fig.org/psr/psr-0/
* @see https://www.php-fig.org/psr/psr-4/
*/
class ClassLoader
{
/** @var \Closure(string):void */
private static $includeFile;
/** @var ?string */
private $vendorDir;
// PSR-4
/**
* @var array[]
* @psalm-var array<string, array<string, int>>
*/
private $prefixLengthsPsr4 = array();
/**
* @var array[]
* @psalm-var array<string, array<int, string>>
*/
private $prefixDirsPsr4 = array();
/**
* @var array[]
* @psalm-var array<string, string>
*/
private $fallbackDirsPsr4 = array();
// PSR-0
/**
* @var array[]
* @psalm-var array<string, array<string, string[]>>
*/
private $prefixesPsr0 = array();
/**
* @var array[]
* @psalm-var array<string, string>
*/
private $fallbackDirsPsr0 = array();
/** @var bool */
private $useIncludePath = false;
/**
* @var string[]
* @psalm-var array<string, string>
*/
private $classMap = array();
/** @var bool */
private $classMapAuthoritative = false;
/**
* @var bool[]
* @psalm-var array<string, bool>
*/
private $missingClasses = array();
/** @var ?string */
private $apcuPrefix;
/**
* @var self[]
*/
private static $registeredLoaders = array();
/**
* @param ?string $vendorDir
*/
public function __construct($vendorDir = null)
{
$this->vendorDir = $vendorDir;
self::initializeIncludeClosure();
}
/**
* @return string[]
*/
public function getPrefixes()
{
if (!empty($this->prefixesPsr0)) {
return call_user_func_array('array_merge', array_values($this->prefixesPsr0));
}
return array();
}
/**
* @return array[]
* @psalm-return array<string, array<int, string>>
*/
public function getPrefixesPsr4()
{
return $this->prefixDirsPsr4;
}
/**
* @return array[]
* @psalm-return array<string, string>
*/
public function getFallbackDirs()
{
return $this->fallbackDirsPsr0;
}
/**
* @return array[]
* @psalm-return array<string, string>
*/
public function getFallbackDirsPsr4()
{
return $this->fallbackDirsPsr4;
}
/**
* @return string[] Array of classname => path
* @psalm-return array<string, string>
*/
public function getClassMap()
{
return $this->classMap;
}
/**
* @param string[] $classMap Class to filename map
* @psalm-param array<string, string> $classMap
*
* @return void
*/
public function addClassMap(array $classMap)
{
if ($this->classMap) {
$this->classMap = array_merge($this->classMap, $classMap);
} else {
$this->classMap = $classMap;
}
}
/**
* Registers a set of PSR-0 directories for a given prefix, either
* appending or prepending to the ones previously set for this prefix.
*
* @param string $prefix The prefix
* @param string[]|string $paths The PSR-0 root directories
* @param bool $prepend Whether to prepend the directories
*
* @return void
*/
public function add($prefix, $paths, $prepend = false)
{
if (!$prefix) {
if ($prepend) {
$this->fallbackDirsPsr0 = array_merge(
(array) $paths,
$this->fallbackDirsPsr0
);
} else {
$this->fallbackDirsPsr0 = array_merge(
$this->fallbackDirsPsr0,
(array) $paths
);
}
return;
}
$first = $prefix[0];
if (!isset($this->prefixesPsr0[$first][$prefix])) {
$this->prefixesPsr0[$first][$prefix] = (array) $paths;
return;
}
if ($prepend) {
$this->prefixesPsr0[$first][$prefix] = array_merge(
(array) $paths,
$this->prefixesPsr0[$first][$prefix]
);
} else {
$this->prefixesPsr0[$first][$prefix] = array_merge(
$this->prefixesPsr0[$first][$prefix],
(array) $paths
);
}
}
/**
* Registers a set of PSR-4 directories for a given namespace, either
* appending or prepending to the ones previously set for this namespace.
*
* @param string $prefix The prefix/namespace, with trailing '\\'
* @param string[]|string $paths The PSR-4 base directories
* @param bool $prepend Whether to prepend the directories
*
* @throws \InvalidArgumentException
*
* @return void
*/
public function addPsr4($prefix, $paths, $prepend = false)
{
if (!$prefix) {
// Register directories for the root namespace.
if ($prepend) {
$this->fallbackDirsPsr4 = array_merge(
(array) $paths,
$this->fallbackDirsPsr4
);
} else {
$this->fallbackDirsPsr4 = array_merge(
$this->fallbackDirsPsr4,
(array) $paths
);
}
} elseif (!isset($this->prefixDirsPsr4[$prefix])) {
// Register directories for a new namespace.
$length = strlen($prefix);
if ('\\' !== $prefix[$length - 1]) {
throw new \InvalidArgumentException("A non-empty PSR-4 prefix must end with a namespace separator.");
}
$this->prefixLengthsPsr4[$prefix[0]][$prefix] = $length;
$this->prefixDirsPsr4[$prefix] = (array) $paths;
} elseif ($prepend) {
// Prepend directories for an already registered namespace.
$this->prefixDirsPsr4[$prefix] = array_merge(
(array) $paths,
$this->prefixDirsPsr4[$prefix]
);
} else {
// Append directories for an already registered namespace.
$this->prefixDirsPsr4[$prefix] = array_merge(
$this->prefixDirsPsr4[$prefix],
(array) $paths
);
}
}
/**
* Registers a set of PSR-0 directories for a given prefix,
* replacing any others previously set for this prefix.
*
* @param string $prefix The prefix
* @param string[]|string $paths The PSR-0 base directories
*
* @return void
*/
public function set($prefix, $paths)
{
if (!$prefix) {
$this->fallbackDirsPsr0 = (array) $paths;
} else {
$this->prefixesPsr0[$prefix[0]][$prefix] = (array) $paths;
}
}
/**
* Registers a set of PSR-4 directories for a given namespace,
* replacing any others previously set for this namespace.
*
* @param string $prefix The prefix/namespace, with trailing '\\'
* @param string[]|string $paths The PSR-4 base directories
*
* @throws \InvalidArgumentException
*
* @return void
*/
public function setPsr4($prefix, $paths)
{
if (!$prefix) {
$this->fallbackDirsPsr4 = (array) $paths;
} else {
$length = strlen($prefix);
if ('\\' !== $prefix[$length - 1]) {
throw new \InvalidArgumentException("A non-empty PSR-4 prefix must end with a namespace separator.");
}
$this->prefixLengthsPsr4[$prefix[0]][$prefix] = $length;
$this->prefixDirsPsr4[$prefix] = (array) $paths;
}
}
/**
* Turns on searching the include path for class files.
*
* @param bool $useIncludePath
*
* @return void
*/
public function setUseIncludePath($useIncludePath)
{
$this->useIncludePath = $useIncludePath;
}
/**
* Can be used to check if the autoloader uses the include path to check
* for classes.
*
* @return bool
*/
public function getUseIncludePath()
{
return $this->useIncludePath;
}
/**
* Turns off searching the prefix and fallback directories for classes
* that have not been registered with the class map.
*
* @param bool $classMapAuthoritative
*
* @return void
*/
public function setClassMapAuthoritative($classMapAuthoritative)
{
$this->classMapAuthoritative = $classMapAuthoritative;
}
/**
* Should class lookup fail if not found in the current class map?
*
* @return bool
*/
public function isClassMapAuthoritative()
{
return $this->classMapAuthoritative;
}
/**
* APCu prefix to use to cache found/not-found classes, if the extension is enabled.
*
* @param string|null $apcuPrefix
*
* @return void
*/
public function setApcuPrefix($apcuPrefix)
{
$this->apcuPrefix = function_exists('apcu_fetch') && filter_var(ini_get('apc.enabled'), FILTER_VALIDATE_BOOLEAN) ? $apcuPrefix : null;
}
/**
* The APCu prefix in use, or null if APCu caching is not enabled.
*
* @return string|null
*/
public function getApcuPrefix()
{
return $this->apcuPrefix;
}
/**
* Registers this instance as an autoloader.
*
* @param bool $prepend Whether to prepend the autoloader or not
*
* @return void
*/
public function register($prepend = false)
{
spl_autoload_register(array($this, 'loadClass'), true, $prepend);
if (null === $this->vendorDir) {
return;
}
if ($prepend) {
self::$registeredLoaders = array($this->vendorDir => $this) + self::$registeredLoaders;
} else {
unset(self::$registeredLoaders[$this->vendorDir]);
self::$registeredLoaders[$this->vendorDir] = $this;
}
}
/**
* Unregisters this instance as an autoloader.
*
* @return void
*/
public function unregister()
{
spl_autoload_unregister(array($this, 'loadClass'));
if (null !== $this->vendorDir) {
unset(self::$registeredLoaders[$this->vendorDir]);
}
}
/**
* Loads the given class or interface.
*
* @param string $class The name of the class
* @return true|null True if loaded, null otherwise
*/
public function loadClass($class)
{
if ($file = $this->findFile($class)) {
$includeFile = self::$includeFile;
$includeFile($file);
return true;
}
return null;
}
/**
* Finds the path to the file where the class is defined.
*
* @param string $class The name of the class
*
* @return string|false The path if found, false otherwise
*/
public function findFile($class)
{
// class map lookup
if (isset($this->classMap[$class])) {
return $this->classMap[$class];
}
if ($this->classMapAuthoritative || isset($this->missingClasses[$class])) {
return false;
}
if (null !== $this->apcuPrefix) {
$file = apcu_fetch($this->apcuPrefix.$class, $hit);
if ($hit) {
return $file;
}
}
$file = $this->findFileWithExtension($class, '.php');
// Search for Hack files if we are running on HHVM
if (false === $file && defined('HHVM_VERSION')) {
$file = $this->findFileWithExtension($class, '.hh');
}
if (null !== $this->apcuPrefix) {
apcu_add($this->apcuPrefix.$class, $file);
}
if (false === $file) {
// Remember that this class does not exist.
$this->missingClasses[$class] = true;
}
return $file;
}
/**
* Returns the currently registered loaders indexed by their corresponding vendor directories.
*
* @return self[]
*/
public static function getRegisteredLoaders()
{
return self::$registeredLoaders;
}
/**
* @param string $class
* @param string $ext
* @return string|false
*/
private function findFileWithExtension($class, $ext)
{
// PSR-4 lookup
$logicalPathPsr4 = strtr($class, '\\', DIRECTORY_SEPARATOR) . $ext;
$first = $class[0];
if (isset($this->prefixLengthsPsr4[$first])) {
$subPath = $class;
while (false !== $lastPos = strrpos($subPath, '\\')) {
$subPath = substr($subPath, 0, $lastPos);
$search = $subPath . '\\';
if (isset($this->prefixDirsPsr4[$search])) {
$pathEnd = DIRECTORY_SEPARATOR . substr($logicalPathPsr4, $lastPos + 1);
foreach ($this->prefixDirsPsr4[$search] as $dir) {
if (file_exists($file = $dir . $pathEnd)) {
return $file;
}
}
}
}
}
// PSR-4 fallback dirs
foreach ($this->fallbackDirsPsr4 as $dir) {
if (file_exists($file = $dir . DIRECTORY_SEPARATOR . $logicalPathPsr4)) {
return $file;
}
}
// PSR-0 lookup
if (false !== $pos = strrpos($class, '\\')) {
// namespaced class name
$logicalPathPsr0 = substr($logicalPathPsr4, 0, $pos + 1)
. strtr(substr($logicalPathPsr4, $pos + 1), '_', DIRECTORY_SEPARATOR);
} else {
// PEAR-like class name
$logicalPathPsr0 = strtr($class, '_', DIRECTORY_SEPARATOR) . $ext;
}
if (isset($this->prefixesPsr0[$first])) {
foreach ($this->prefixesPsr0[$first] as $prefix => $dirs) {
if (0 === strpos($class, $prefix)) {
foreach ($dirs as $dir) {
if (file_exists($file = $dir . DIRECTORY_SEPARATOR . $logicalPathPsr0)) {
return $file;
}
}
}
}
}
// PSR-0 fallback dirs
foreach ($this->fallbackDirsPsr0 as $dir) {
if (file_exists($file = $dir . DIRECTORY_SEPARATOR . $logicalPathPsr0)) {
return $file;
}
}
// PSR-0 include paths.
if ($this->useIncludePath && $file = stream_resolve_include_path($logicalPathPsr0)) {
return $file;
}
return false;
}
/**
* @return void
*/
private static function initializeIncludeClosure()
{
if (self::$includeFile !== null) {
return;
}
/**
* Scope isolated include.
*
* Prevents access to $this/self from included files.
*
* @param string $file
* @return void
*/
self::$includeFile = \Closure::bind(static function($file) {
include $file;
}, null, null);
}
}
+352
View File
@@ -0,0 +1,352 @@
<?php
/*
* This file is part of Composer.
*
* (c) Nils Adermann <naderman@naderman.de>
* Jordi Boggiano <j.boggiano@seld.be>
*
* For the full copyright and license information, please view the LICENSE
* file that was distributed with this source code.
*/
namespace Composer;
use Composer\Autoload\ClassLoader;
use Composer\Semver\VersionParser;
/**
* This class is copied in every Composer installed project and available to all
*
* See also https://getcomposer.org/doc/07-runtime.md#installed-versions
*
* To require its presence, you can require `composer-runtime-api ^2.0`
*
* @final
*/
class InstalledVersions
{
/**
* @var mixed[]|null
* @psalm-var array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>}|array{}|null
*/
private static $installed;
/**
* @var bool|null
*/
private static $canGetVendors;
/**
* @var array[]
* @psalm-var array<string, array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>}>
*/
private static $installedByVendor = array();
/**
* Returns a list of all package names which are present, either by being installed, replaced or provided
*
* @return string[]
* @psalm-return list<string>
*/
public static function getInstalledPackages()
{
$packages = array();
foreach (self::getInstalled() as $installed) {
$packages[] = array_keys($installed['versions']);
}
if (1 === \count($packages)) {
return $packages[0];
}
return array_keys(array_flip(\call_user_func_array('array_merge', $packages)));
}
/**
* Returns a list of all package names with a specific type e.g. 'library'
*
* @param string $type
* @return string[]
* @psalm-return list<string>
*/
public static function getInstalledPackagesByType($type)
{
$packagesByType = array();
foreach (self::getInstalled() as $installed) {
foreach ($installed['versions'] as $name => $package) {
if (isset($package['type']) && $package['type'] === $type) {
$packagesByType[] = $name;
}
}
}
return $packagesByType;
}
/**
* Checks whether the given package is installed
*
* This also returns true if the package name is provided or replaced by another package
*
* @param string $packageName
* @param bool $includeDevRequirements
* @return bool
*/
public static function isInstalled($packageName, $includeDevRequirements = true)
{
foreach (self::getInstalled() as $installed) {
if (isset($installed['versions'][$packageName])) {
return $includeDevRequirements || empty($installed['versions'][$packageName]['dev_requirement']);
}
}
return false;
}
/**
* Checks whether the given package satisfies a version constraint
*
* e.g. If you want to know whether version 2.3+ of package foo/bar is installed, you would call:
*
* Composer\InstalledVersions::satisfies(new VersionParser, 'foo/bar', '^2.3')
*
* @param VersionParser $parser Install composer/semver to have access to this class and functionality
* @param string $packageName
* @param string|null $constraint A version constraint to check for, if you pass one you have to make sure composer/semver is required by your package
* @return bool
*/
public static function satisfies(VersionParser $parser, $packageName, $constraint)
{
$constraint = $parser->parseConstraints($constraint);
$provided = $parser->parseConstraints(self::getVersionRanges($packageName));
return $provided->matches($constraint);
}
/**
* Returns a version constraint representing all the range(s) which are installed for a given package
*
* It is easier to use this via isInstalled() with the $constraint argument if you need to check
* whether a given version of a package is installed, and not just whether it exists
*
* @param string $packageName
* @return string Version constraint usable with composer/semver
*/
public static function getVersionRanges($packageName)
{
foreach (self::getInstalled() as $installed) {
if (!isset($installed['versions'][$packageName])) {
continue;
}
$ranges = array();
if (isset($installed['versions'][$packageName]['pretty_version'])) {
$ranges[] = $installed['versions'][$packageName]['pretty_version'];
}
if (array_key_exists('aliases', $installed['versions'][$packageName])) {
$ranges = array_merge($ranges, $installed['versions'][$packageName]['aliases']);
}
if (array_key_exists('replaced', $installed['versions'][$packageName])) {
$ranges = array_merge($ranges, $installed['versions'][$packageName]['replaced']);
}
if (array_key_exists('provided', $installed['versions'][$packageName])) {
$ranges = array_merge($ranges, $installed['versions'][$packageName]['provided']);
}
return implode(' || ', $ranges);
}
throw new \OutOfBoundsException('Package "' . $packageName . '" is not installed');
}
/**
* @param string $packageName
* @return string|null If the package is being replaced or provided but is not really installed, null will be returned as version, use satisfies or getVersionRanges if you need to know if a given version is present
*/
public static function getVersion($packageName)
{
foreach (self::getInstalled() as $installed) {
if (!isset($installed['versions'][$packageName])) {
continue;
}
if (!isset($installed['versions'][$packageName]['version'])) {
return null;
}
return $installed['versions'][$packageName]['version'];
}
throw new \OutOfBoundsException('Package "' . $packageName . '" is not installed');
}
/**
* @param string $packageName
* @return string|null If the package is being replaced or provided but is not really installed, null will be returned as version, use satisfies or getVersionRanges if you need to know if a given version is present
*/
public static function getPrettyVersion($packageName)
{
foreach (self::getInstalled() as $installed) {
if (!isset($installed['versions'][$packageName])) {
continue;
}
if (!isset($installed['versions'][$packageName]['pretty_version'])) {
return null;
}
return $installed['versions'][$packageName]['pretty_version'];
}
throw new \OutOfBoundsException('Package "' . $packageName . '" is not installed');
}
/**
* @param string $packageName
* @return string|null If the package is being replaced or provided but is not really installed, null will be returned as reference
*/
public static function getReference($packageName)
{
foreach (self::getInstalled() as $installed) {
if (!isset($installed['versions'][$packageName])) {
continue;
}
if (!isset($installed['versions'][$packageName]['reference'])) {
return null;
}
return $installed['versions'][$packageName]['reference'];
}
throw new \OutOfBoundsException('Package "' . $packageName . '" is not installed');
}
/**
* @param string $packageName
* @return string|null If the package is being replaced or provided but is not really installed, null will be returned as install path. Packages of type metapackages also have a null install path.
*/
public static function getInstallPath($packageName)
{
foreach (self::getInstalled() as $installed) {
if (!isset($installed['versions'][$packageName])) {
continue;
}
return isset($installed['versions'][$packageName]['install_path']) ? $installed['versions'][$packageName]['install_path'] : null;
}
throw new \OutOfBoundsException('Package "' . $packageName . '" is not installed');
}
/**
* @return array
* @psalm-return array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}
*/
public static function getRootPackage()
{
$installed = self::getInstalled();
return $installed[0]['root'];
}
/**
* Returns the raw installed.php data for custom implementations
*
* @deprecated Use getAllRawData() instead which returns all datasets for all autoloaders present in the process. getRawData only returns the first dataset loaded, which may not be what you expect.
* @return array[]
* @psalm-return array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>}
*/
public static function getRawData()
{
@trigger_error('getRawData only returns the first dataset loaded, which may not be what you expect. Use getAllRawData() instead which returns all datasets for all autoloaders present in the process.', E_USER_DEPRECATED);
if (null === self::$installed) {
// only require the installed.php file if this file is loaded from its dumped location,
// and not from its source location in the composer/composer package, see https://github.com/composer/composer/issues/9937
if (substr(__DIR__, -8, 1) !== 'C') {
self::$installed = include __DIR__ . '/installed.php';
} else {
self::$installed = array();
}
}
return self::$installed;
}
/**
* Returns the raw data of all installed.php which are currently loaded for custom implementations
*
* @return array[]
* @psalm-return list<array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>}>
*/
public static function getAllRawData()
{
return self::getInstalled();
}
/**
* Lets you reload the static array from another file
*
* This is only useful for complex integrations in which a project needs to use
* this class but then also needs to execute another project's autoloader in process,
* and wants to ensure both projects have access to their version of installed.php.
*
* A typical case would be PHPUnit, where it would need to make sure it reads all
* the data it needs from this class, then call reload() with
* `require $CWD/vendor/composer/installed.php` (or similar) as input to make sure
* the project in which it runs can then also use this class safely, without
* interference between PHPUnit's dependencies and the project's dependencies.
*
* @param array[] $data A vendor/composer/installed.php data set
* @return void
*
* @psalm-param array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>} $data
*/
public static function reload($data)
{
self::$installed = $data;
self::$installedByVendor = array();
}
/**
* @return array[]
* @psalm-return list<array{root: array{name: string, pretty_version: string, version: string, reference: string|null, type: string, install_path: string, aliases: string[], dev: bool}, versions: array<string, array{pretty_version?: string, version?: string, reference?: string|null, type?: string, install_path?: string, aliases?: string[], dev_requirement: bool, replaced?: string[], provided?: string[]}>}>
*/
private static function getInstalled()
{
if (null === self::$canGetVendors) {
self::$canGetVendors = method_exists('Composer\Autoload\ClassLoader', 'getRegisteredLoaders');
}
$installed = array();
if (self::$canGetVendors) {
foreach (ClassLoader::getRegisteredLoaders() as $vendorDir => $loader) {
if (isset(self::$installedByVendor[$vendorDir])) {
$installed[] = self::$installedByVendor[$vendorDir];
} elseif (is_file($vendorDir.'/composer/installed.php')) {
$installed[] = self::$installedByVendor[$vendorDir] = require $vendorDir.'/composer/installed.php';
if (null === self::$installed && strtr($vendorDir.'/composer', '\\', '/') === strtr(__DIR__, '\\', '/')) {
self::$installed = $installed[count($installed) - 1];
}
}
}
}
if (null === self::$installed) {
// only require the installed.php file if this file is loaded from its dumped location,
// and not from its source location in the composer/composer package, see https://github.com/composer/composer/issues/9937
if (substr(__DIR__, -8, 1) !== 'C') {
self::$installed = require __DIR__ . '/installed.php';
} else {
self::$installed = array();
}
}
$installed[] = self::$installed;
return $installed;
}
}
+21
View File
@@ -0,0 +1,21 @@
Copyright (c) Nils Adermann, Jordi Boggiano
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is furnished
to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
+11
View File
@@ -0,0 +1,11 @@
<?php
// autoload_classmap.php @generated by Composer
$vendorDir = dirname(__DIR__);
$baseDir = dirname($vendorDir);
return array(
'Composer\\InstalledVersions' => $vendorDir . '/composer/InstalledVersions.php',
'Grav\\Plugin\\TNTSearchPlugin' => $baseDir . '/tntsearch.php',
);
+10
View File
@@ -0,0 +1,10 @@
<?php
// autoload_files.php @generated by Composer
$vendorDir = dirname(__DIR__);
$baseDir = dirname($vendorDir);
return array(
'290dd4ba42f11019134caca05dbefe3f' => $vendorDir . '/teamtnt/tntsearch/helper/helpers.php',
);
@@ -0,0 +1,9 @@
<?php
// autoload_namespaces.php @generated by Composer
$vendorDir = dirname(__DIR__);
$baseDir = dirname($vendorDir);
return array(
);
+12
View File
@@ -0,0 +1,12 @@
<?php
// autoload_psr4.php @generated by Composer
$vendorDir = dirname(__DIR__);
$baseDir = dirname($vendorDir);
return array(
'TeamTNT\\TNTSearch\\' => array($vendorDir . '/teamtnt/tntsearch/src'),
'Grav\\Plugin\\TNTSearch\\' => array($baseDir . '/classes'),
'Grav\\Plugin\\Console\\' => array($baseDir . '/cli'),
);
+50
View File
@@ -0,0 +1,50 @@
<?php
// autoload_real.php @generated by Composer
class ComposerAutoloaderInit6693564509f9a3fa6ed2c7bf76fdb017
{
private static $loader;
public static function loadClassLoader($class)
{
if ('Composer\Autoload\ClassLoader' === $class) {
require __DIR__ . '/ClassLoader.php';
}
}
/**
* @return \Composer\Autoload\ClassLoader
*/
public static function getLoader()
{
if (null !== self::$loader) {
return self::$loader;
}
require __DIR__ . '/platform_check.php';
spl_autoload_register(array('ComposerAutoloaderInit6693564509f9a3fa6ed2c7bf76fdb017', 'loadClassLoader'), true, true);
self::$loader = $loader = new \Composer\Autoload\ClassLoader(\dirname(__DIR__));
spl_autoload_unregister(array('ComposerAutoloaderInit6693564509f9a3fa6ed2c7bf76fdb017', 'loadClassLoader'));
require __DIR__ . '/autoload_static.php';
call_user_func(\Composer\Autoload\ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017::getInitializer($loader));
$loader->register(true);
$filesToLoad = \Composer\Autoload\ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017::$files;
$requireFile = \Closure::bind(static function ($fileIdentifier, $file) {
if (empty($GLOBALS['__composer_autoload_files'][$fileIdentifier])) {
$GLOBALS['__composer_autoload_files'][$fileIdentifier] = true;
require $file;
}
}, null, null);
foreach ($filesToLoad as $fileIdentifier => $file) {
$requireFile($fileIdentifier, $file);
}
return $loader;
}
}
+54
View File
@@ -0,0 +1,54 @@
<?php
// autoload_static.php @generated by Composer
namespace Composer\Autoload;
class ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017
{
public static $files = array (
'290dd4ba42f11019134caca05dbefe3f' => __DIR__ . '/..' . '/teamtnt/tntsearch/helper/helpers.php',
);
public static $prefixLengthsPsr4 = array (
'T' =>
array (
'TeamTNT\\TNTSearch\\' => 18,
),
'G' =>
array (
'Grav\\Plugin\\TNTSearch\\' => 22,
'Grav\\Plugin\\Console\\' => 20,
),
);
public static $prefixDirsPsr4 = array (
'TeamTNT\\TNTSearch\\' =>
array (
0 => __DIR__ . '/..' . '/teamtnt/tntsearch/src',
),
'Grav\\Plugin\\TNTSearch\\' =>
array (
0 => __DIR__ . '/../..' . '/classes',
),
'Grav\\Plugin\\Console\\' =>
array (
0 => __DIR__ . '/../..' . '/cli',
),
);
public static $classMap = array (
'Composer\\InstalledVersions' => __DIR__ . '/..' . '/composer/InstalledVersions.php',
'Grav\\Plugin\\TNTSearchPlugin' => __DIR__ . '/../..' . '/tntsearch.php',
);
public static function getInitializer(ClassLoader $loader)
{
return \Closure::bind(function () use ($loader) {
$loader->prefixLengthsPsr4 = ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017::$prefixLengthsPsr4;
$loader->prefixDirsPsr4 = ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017::$prefixDirsPsr4;
$loader->classMap = ComposerStaticInit6693564509f9a3fa6ed2c7bf76fdb017::$classMap;
}, null, ClassLoader::class);
}
}
+87
View File
@@ -0,0 +1,87 @@
{
"packages": [
{
"name": "teamtnt/tntsearch",
"version": "v2.9.0",
"version_normalized": "2.9.0.0",
"source": {
"type": "git",
"url": "https://github.com/teamtnt/tntsearch.git",
"reference": "ccedae0cfe21f7831f2dd1f973cf8904dad42d8d"
},
"dist": {
"type": "zip",
"url": "https://api.github.com/repos/teamtnt/tntsearch/zipball/ccedae0cfe21f7831f2dd1f973cf8904dad42d8d",
"reference": "ccedae0cfe21f7831f2dd1f973cf8904dad42d8d",
"shasum": ""
},
"require": {
"ext-mbstring": "*",
"ext-pdo_sqlite": "*",
"ext-sqlite3": "*",
"php": "~7.1|^8"
},
"require-dev": {
"phpunit/phpunit": "7.*|8.*|9.*",
"symfony/var-dumper": "^4|^5.2"
},
"time": "2022-02-22T10:35:34+00:00",
"type": "library",
"installation-source": "dist",
"autoload": {
"files": [
"helper/helpers.php"
],
"psr-4": {
"TeamTNT\\TNTSearch\\": "src"
}
},
"notification-url": "https://packagist.org/downloads/",
"license": [
"MIT"
],
"authors": [
{
"name": "Nenad Tičarić",
"email": "nticaric@gmail.com",
"homepage": "http://www.tntstudio.us",
"role": "Developer"
}
],
"description": "A fully featured full text search engine written in PHP",
"homepage": "https://github.com/teamtnt/tntsearch",
"keywords": [
"Fuzzy search",
"bm25",
"fulltext",
"geosearch",
"search",
"stemming",
"teamtnt",
"text classification",
"tntsearch"
],
"support": {
"issues": "https://github.com/teamtnt/tntsearch/issues",
"source": "https://github.com/teamtnt/tntsearch/tree/v2.9.0"
},
"funding": [
{
"url": "https://ko-fi.com/nticaric",
"type": "ko_fi"
},
{
"url": "https://opencollective.com/tntsearch",
"type": "open_collective"
},
{
"url": "https://www.patreon.com/nticaric",
"type": "patreon"
}
],
"install-path": "../teamtnt/tntsearch"
}
],
"dev": true,
"dev-package-names": []
}
+32
View File
@@ -0,0 +1,32 @@
<?php return array(
'root' => array(
'name' => 'trilbymedia/grav-plugin-tntsearch',
'pretty_version' => 'dev-develop',
'version' => 'dev-develop',
'reference' => '60562d62856c114f23c183f7873fe1c809f4c7b5',
'type' => 'grav-plugin',
'install_path' => __DIR__ . '/../../',
'aliases' => array(),
'dev' => true,
),
'versions' => array(
'teamtnt/tntsearch' => array(
'pretty_version' => 'v2.9.0',
'version' => '2.9.0.0',
'reference' => 'ccedae0cfe21f7831f2dd1f973cf8904dad42d8d',
'type' => 'library',
'install_path' => __DIR__ . '/../teamtnt/tntsearch',
'aliases' => array(),
'dev_requirement' => false,
),
'trilbymedia/grav-plugin-tntsearch' => array(
'pretty_version' => 'dev-develop',
'version' => 'dev-develop',
'reference' => '60562d62856c114f23c183f7873fe1c809f4c7b5',
'type' => 'grav-plugin',
'install_path' => __DIR__ . '/../../',
'aliases' => array(),
'dev_requirement' => false,
),
),
);
+26
View File
@@ -0,0 +1,26 @@
<?php
// platform_check.php @generated by Composer
$issues = array();
if (!(PHP_VERSION_ID >= 70103)) {
$issues[] = 'Your Composer dependencies require a PHP version ">= 7.1.3". You are running ' . PHP_VERSION . '.';
}
if ($issues) {
if (!headers_sent()) {
header('HTTP/1.1 500 Internal Server Error');
}
if (!ini_get('display_errors')) {
if (PHP_SAPI === 'cli' || PHP_SAPI === 'phpdbg') {
fwrite(STDERR, 'Composer detected issues in your platform:' . PHP_EOL.PHP_EOL . implode(PHP_EOL, $issues) . PHP_EOL.PHP_EOL);
} elseif (!headers_sent()) {
echo 'Composer detected issues in your platform:' . PHP_EOL.PHP_EOL . str_replace('You are running '.PHP_VERSION.'.', '', implode(PHP_EOL, $issues)) . PHP_EOL.PHP_EOL;
}
}
trigger_error(
'Composer detected issues in your platform: ' . implode(' ', $issues),
E_USER_ERROR
);
}
@@ -0,0 +1,3 @@
open_collective: tntsearch
patreon: nticaric
ko_fi: nticaric
@@ -0,0 +1,18 @@
# Number of days of inactivity before an issue becomes stale
daysUntilStale: 240
# Number of days of inactivity before a stale issue is closed
daysUntilClose: 7
# Issues with these labels will never be considered stale
exemptLabels:
- pinned
- security
- PR
# Label to use when marking an issue as stale
staleLabel: wontfix
# Comment to post when marking an issue as stale. Set to `false` to disable
markComment: >
This issue has been automatically marked as stale because it has not had
recent activity. It will be closed if no further activity occurs. Thank you
for your contributions.
# Comment to post when closing a stale issue. Set to `false` to disable
closeComment: false
@@ -0,0 +1,8 @@
.idea/*
vendor
examples
.DS_Store
composer.lock
coverage
tests/_files/*.index
.phpunit.result.cache
+22
View File
@@ -0,0 +1,22 @@
language: php
php:
- 7.1
- 7.2
- 7.3
- 7.4
- 8.0
addons:
code_climate:
repo_token: e43f1f89afb5a2f6acfaea42a6a9ebd8d33538208fafa8636826c173b3f7ec26
script:
- vendor/bin/phpunit
before_script:
- composer self-update
- composer install
after_script:
- vendor/bin/test-reporter
+22
View File
@@ -0,0 +1,22 @@
# Changelog
All Notable changes to `tntsearch` will be documented in this file.
Updates should follow the [Keep a CHANGELOG](http://keepachangelog.com/) principles.
## NEXT - YYYY-MM-DD
### Added
- Nothing
### Deprecated
- Nothing
### Fixed
- Nothing
### Removed
- Nothing
### Security
- Nothing
@@ -0,0 +1,46 @@
# Contributor Covenant Code of Conduct
## Our Pledge
In the interest of fostering an open and welcoming environment, we as contributors and maintainers pledge to making participation in our project and our community a harassment-free experience for everyone, regardless of age, body size, disability, ethnicity, gender identity and expression, level of experience, nationality, personal appearance, race, religion, or sexual identity and orientation.
## Our Standards
Examples of behavior that contributes to creating a positive environment include:
* Using welcoming and inclusive language
* Being respectful of differing viewpoints and experiences
* Gracefully accepting constructive criticism
* Focusing on what is best for the community
* Showing empathy towards other community members
Examples of unacceptable behavior by participants include:
* The use of sexualized language or imagery and unwelcome sexual attention or advances
* Trolling, insulting/derogatory comments, and personal or political attacks
* Public or private harassment
* Publishing others' private information, such as a physical or electronic address, without explicit permission
* Other conduct which could reasonably be considered inappropriate in a professional setting
## Our Responsibilities
Project maintainers are responsible for clarifying the standards of acceptable behavior and are expected to take appropriate and fair corrective action in response to any instances of unacceptable behavior.
Project maintainers have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, or to ban temporarily or permanently any contributor for other behaviors that they deem inappropriate, threatening, offensive, or harmful.
## Scope
This Code of Conduct applies both within project spaces and in public spaces when an individual is representing the project or its community. Examples of representing a project or community include using an official project e-mail address, posting via an official social media account, or acting as an appointed representative at an online or offline event. Representation of a project may be further defined and clarified by project maintainers.
## Enforcement
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported by contacting the project team at info@tntstudio.hr. The project team will review and investigate all complaints, and will respond in a way that it deems appropriate to the circumstances. The project team is obligated to maintain confidentiality with regard to the reporter of an incident. Further details of specific enforcement policies may be posted separately.
Project maintainers who do not follow or enforce the Code of Conduct in good faith may face temporary or permanent repercussions as determined by other members of the project's leadership.
## Attribution
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, available at [http://contributor-covenant.org/version/1/4][version]
[homepage]: http://contributor-covenant.org
[version]: http://contributor-covenant.org/version/1/4/
+22
View File
@@ -0,0 +1,22 @@
# Contributor Code of Conduct
As contributors and maintainers of this project, and in the interest of fostering an open and welcoming community, we pledge to respect all people who contribute through reporting issues, posting feature requests, updating documentation, submitting pull requests or patches, and other activities.
We are committed to making participation in this project a harassment-free experience for everyone, regardless of level of experience, gender, gender identity and expression, sexual orientation, disability, personal appearance, body size, race, ethnicity, age, religion, or nationality.
Examples of unacceptable behavior by participants include:
* The use of sexualized language or imagery
* Personal attacks
* Trolling or insulting/derogatory comments
* Public or private harassment
* Publishing other's private information, such as physical or electronic addresses, without explicit permission
* Other unethical or unprofessional conduct.
Project maintainers have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct. By adopting this Code of Conduct, project maintainers commit themselves to fairly and consistently applying these principles to every aspect of managing this project. Project maintainers who do not follow or enforce the Code of Conduct may be permanently removed from the project team.
This code of conduct applies both within project spaces and in public spaces when an individual is representing the project or its community in a direct capacity. Personal views, beliefs and values of individuals do not necessarily reflect those of the organisation or affiliated individuals and organisations.
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported by opening an issue or contacting one or more of the project maintainers.
This Code of Conduct is adapted from the [Contributor Covenant](http://contributor-covenant.org), version 1.2.0, available at [http://contributor-covenant.org/version/1/2/0/](http://contributor-covenant.org/version/1/2/0/)
@@ -0,0 +1,32 @@
# Contributing
Contributions are **welcome** and will be fully **credited**.
We accept contributions via Pull Requests on [Github](https://github.com/teamtnt/tntsearch).
## Pull Requests
- **[PSR-2 Coding Standard](https://github.com/php-fig/fig-standards/blob/master/accepted/PSR-2-coding-style-guide.md)** - The easiest way to apply the conventions is to install [PHP Code Sniffer](http://pear.php.net/package/PHP_CodeSniffer).
- **Add tests!** - Your patch won't be accepted if it doesn't have tests.
- **Document any change in behaviour** - Make sure the `README.md` and any other relevant documentation are kept up-to-date.
- **Consider our release cycle** - We try to follow [SemVer v2.0.0](http://semver.org/). Randomly breaking public APIs is not an option.
- **Create feature branches** - Don't ask us to pull from your master branch.
- **One pull request per feature** - If you want to do more than one thing, send multiple pull requests.
- **Send coherent history** - Make sure each individual commit in your pull request is meaningful. If you had to make multiple intermediate commits while developing, please [squash them](http://www.git-scm.com/book/en/v2/Git-Tools-Rewriting-History#Changing-Multiple-Commit-Messages) before submitting.
## Running Tests
``` bash
$ composer test
```
**Happy coding**!
+21
View File
@@ -0,0 +1,21 @@
# The MIT License (MIT)
Copyright (c) 2016 Nenad Tičarić <nticaric@gmail.com>
> Permission is hereby granted, free of charge, to any person obtaining a copy
> of this software and associated documentation files (the "Software"), to deal
> in the Software without restriction, including without limitation the rights
> to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
> copies of the Software, and to permit persons to whom the Software is
> furnished to do so, subject to the following conditions:
>
> The above copyright notice and this permission notice shall be included in
> all copies or substantial portions of the Software.
>
> THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
> IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
> FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
> AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
> LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
> OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
> THE SOFTWARE.
+9
View File
@@ -0,0 +1,9 @@
# PS4Ware
TNTSearch is PS4Ware: it's free to use, but if it makes to production
we'd appreciate a PS4 game.
### [Helm und Walter Team](https://helmundwalter.de/)
![The Long Dark](https://user-images.githubusercontent.com/824840/66302347-0e8af800-e8f9-11e9-96d2-4bbf58532f34.png)
+380
View File
@@ -0,0 +1,380 @@
[![Latest Version on Packagist][ico-version]][link-packagist]
[![Total Downloads][ico-downloads]][link-downloads]
[![Software License][ico-license]](LICENSE.md)
[![Build Status](https://img.shields.io/travis/teamtnt/tntsearch/master.svg?style=flat-square)](https://travis-ci.org/teamtnt/tntsearch)
[![Slack Status](https://img.shields.io/badge/slack-chat-E01563.svg?style=flat-square)](https://tntsearch.slack.com)
![TNTSearch](https://i.imgur.com/aYKsNYv.png)
# TNTSearch
TNTSearch is a full-text search (FTS) engine written entirely in PHP. A simple configuration allows you to add an amazing search experience in just minutes. Features include:
* Fuzzy search
* Search as you type
* Geo-search
* Text classification
* Stemming
* Custom tokenizers
* Bm25 ranking algorithm
* Boolean search
* Result highlighting
* Dynamic index updates (no need to reindex each time)
* Easily deployable via Packagist.org
We also created some demo pages that show tolerant retrieval with n-grams in action.
The package has a bunch of helper functions like Jaro-Winkler and Cosine similarity for distance calculations. It supports stemming for English, Croatian, Arabic, Italian, Russian, Portuguese and Ukrainian. If the built-in stemmers aren't enough, the engine lets you easily plugin any compatible snowball stemmer. Some forks of the package even support Chinese. And please contribute other languages!
Unlike many other engines, the index can be easily updated without doing a reindex or using deltas.
**View** [online demo](http://tntsearch.tntstudio.us/) &nbsp;|&nbsp; **Follow us** on
[Twitter](https://twitter.com/tntstudiohr),
or [Facebook](https://www.facebook.com/tntstudiohr) &nbsp;|&nbsp;
**Visit our sponsors**:
<p align="center">
<a href="https://m.do.co/c/ddfc227b7d18" target="_blank">
<img src="https://images.prismic.io/www-static/49aa0a09-06d2-4bba-ad20-4bcbe56ac507_logo.png?auto=compress,format" width="196.5" height="32">
</a>
</p>
---
## Demo
* [TV Shows Search](http://tntsearch.tntstudio.us/)
* [PHPUnit Documentation Search](http://phpunit.tntstudio.us)
* [City Search with n-grams](http://cities.tnt.studio/)
## Tutorials
* [Solving the search problem with Laravel and TNTSearch](https://tnt.studio/solving-the-search-problem-with-laravel-and-tntsearch)
* [Searching for Users with Laravel Scout and TNTSearch](https://tnt.studio/searching-for-users-with-laravel-scout-and-tntsearch)
## Premium products
If you're using TNT Search and finding it useful, take a look at our premium analytics tool:
[<img src="https://i.imgur.com/ujagviB.png" width="420px" />](https://analytics.tnt.studio)
## Support us on Open Collective
- [TNTSearch](https://opencollective.com/tntsearch)
## Installation
The easiest way to install TNTSearch is via [composer](http://getcomposer.org/):
```
composer require teamtnt/tntsearch
```
## Requirements
Before you proceed, make sure your server meets the following requirements:
* PHP >= 7.1
* PDO PHP Extension
* SQLite PHP Extension
* mbstring PHP Extension
## Examples
### Creating an index
In order to be able to make full text search queries, you have to create an index.
Usage:
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig([
'driver' => 'mysql',
'host' => 'localhost',
'database' => 'dbname',
'username' => 'user',
'password' => 'pass',
'storage' => '/var/www/tntsearch/examples/',
'stemmer' => \TeamTNT\TNTSearch\Stemmer\PorterStemmer::class//optional
]);
$indexer = $tnt->createIndex('name.index');
$indexer->query('SELECT id, article FROM articles;');
//$indexer->setLanguage('german');
$indexer->run();
```
Important: "storage" settings marks the folder where all of your indexes
will be saved so make sure to have permission to write to this folder otherwise
you might expect the following exception thrown:
* [PDOException] SQLSTATE[HY000] [14] unable to open database file *
Note: If your primary key is different than `id` set it like:
```php
$indexer->setPrimaryKey('article_id');
```
### Making the primary key searchable
By default, the primary key isn't searchable. If you want to make it searchable, simply run:
```php
$indexer->includePrimaryKey();
```
### Searching
Searching for a phrase or keyword is trivial:
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig($config);
$tnt->selectIndex("name.index");
$res = $tnt->search("This is a test search", 12);
print_r($res); //returns an array of 12 document ids that best match your query
// to display the results you need an additional query against your application database
// SELECT * FROM articles WHERE id IN $res ORDER BY FIELD(id, $res);
```
The ORDER BY FIELD clause is important, otherwise the database engine will not return
the results in the required order.
### Boolean Search
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig($config);
$tnt->selectIndex("name.index");
//this will return all documents that have romeo in it but not juliet
$res = $tnt->searchBoolean("romeo -juliet");
//returns all documents that have romeo or hamlet in it
$res = $tnt->searchBoolean("romeo or hamlet");
//returns all documents that have either romeo AND juliet or prince AND hamlet
$res = $tnt->searchBoolean("(romeo juliet) or (prince hamlet)");
```
### Fuzzy Search
The fuzziness can be tweaked by setting the following member variables:
```php
public $fuzzy_prefix_length = 2;
public $fuzzy_max_expansions = 50;
public $fuzzy_distance = 2; //represents the Levenshtein distance;
```
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig($config);
$tnt->selectIndex("name.index");
$tnt->fuzziness = true;
//when the fuzziness flag is set to true, the keyword juleit will return
//documents that match the word juliet, the default Levenshtein distance is 2
$res = $tnt->search("juleit");
```
## Updating the index
Once you created an index, you don't need to reindex it each time you make some changes
to your document collection. TNTSearch supports dynamic index updates.
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig($config);
$tnt->selectIndex("name.index");
$index = $tnt->getIndex();
//to insert a new document to the index
$index->insert(['id' => '11', 'title' => 'new title', 'article' => 'new article']);
//to update an existing document
$index->update(11, ['id' => '11', 'title' => 'updated title', 'article' => 'updated article']);
//to delete the document from index
$index->delete(12);
```
## Custom Tokenizer
First, create your own Tokenizer class. It should extend AbstractTokenizer class, define
word split $pattern value and must implement TokenizerInterface:
``` php
use TeamTNT\TNTSearch\Support\AbstractTokenizer;
use TeamTNT\TNTSearch\Support\TokenizerInterface;
class SomeTokenizer extends AbstractTokenizer implements TokenizerInterface
{
static protected $pattern = '/[\s,\.]+/';
public function tokenize($text) {
return preg_split($this->getPattern(), strtolower($text), -1, PREG_SPLIT_NO_EMPTY);
}
}
```
This tokenizer will split words using spaces, commas and periods.
After you have the tokenizer ready, you should pass it to `TNTIndexer` via `setTokenizer` method.
``` php
$someTokenizer = new SomeTokenizer;
$indexer = new TNTIndexer;
$indexer->setTokenizer($someTokenizer);
```
Another way would be to pass the tokenizer via config:
```php
use TeamTNT\TNTSearch\TNTSearch;
$tnt = new TNTSearch;
$tnt->loadConfig([
'driver' => 'mysql',
'host' => 'localhost',
'database' => 'dbname',
'username' => 'user',
'password' => 'pass',
'storage' => '/var/www/tntsearch/examples/',
'stemmer' => \TeamTNT\TNTSearch\Stemmer\PorterStemmer::class//optional,
'tokenizer' => \TeamTNT\TNTSearch\Support\SomeTokenizer::class
]);
$indexer = $tnt->createIndex('name.index');
$indexer->query('SELECT id, article FROM articles;');
$indexer->run();
```
## Geo Search
### Indexing
```php
$candyShopIndexer = new TNTGeoIndexer;
$candyShopIndexer->loadConfig($config);
$candyShopIndexer->createIndex('candyShops.index');
$candyShopIndexer->query('SELECT id, longitude, latitude FROM candy_shops;');
$candyShopIndexer->run();
```
### Searching
```php
$currentLocation = [
'longitude' => 11.576124,
'latitude' => 48.137154
];
$distance = 2; //km
$candyShopIndex = new TNTGeoSearch();
$candyShopIndex->loadConfig($config);
$candyShopIndex->selectIndex('candyShops.index');
$candyShops = $candyShopIndex->findNearest($currentLocation, $distance, 10);
```
## Classification
```php
use TeamTNT\TNTSearch\Classifier\TNTClassifier;
$classifier = new TNTClassifier();
$classifier->learn("A great game", "Sports");
$classifier->learn("The election was over", "Not sports");
$classifier->learn("Very clean match", "Sports");
$classifier->learn("A clean but forgettable game", "Sports");
$guess = $classifier->predict("It was a close election");
var_dump($guess['label']); //returns "Not sports"
```
### Saving the classifier
```php
$classifier->save('sports.cls');
```
### Loading the classifier
```php
$classifier = new TNTClassifier();
$classifier->load('sports.cls');
```
## Drivers
* [TNTSearch Driver for Laravel Scout](https://github.com/teamtnt/laravel-scout-tntsearch-driver)
## PS4Ware
You're free to use this package, but if it makes it to your production environment, we would highly appreciate you sending us a PS4 game of your choice. This way you support us to further develop and add new features.
Our address is: TNT Studio, Sv. Mateja 19, 10010 Zagreb, Croatia.
We'll publish all received games [here][link-ps4ware]
[link-ps4ware]: https://github.com/teamtnt/tntsearch/blob/master/PS4Ware.md
## Support [![OpenCollective](https://opencollective.com/tntsearch/backers/badge.svg)](#backers) [![OpenCollective](https://opencollective.com/tntsearch/sponsors/badge.svg)](#sponsors)
<a href='https://ko-fi.com/O4O3K2R9' target='_blank'><img height='36' style='border:0px;height:36px;' src='https://az743702.vo.msecnd.net/cdn/kofi4.png?v=0' border='0' alt='Buy Me a Coffee at ko-fi.com' /></a>
### Backers
Support us with a monthly donation and help us continue our activities. [[Become a backer](https://opencollective.com/tntsearch#backer)]
## Sponsors
Become a sponsor and get your logo on our README on Github with a link to your site. [[Become a sponsor](https://opencollective.com/tntsearch#sponsor)]
## Credits
- [Nenad Tičarić][link-author]
- [All Contributors][link-contributors]
## License
The MIT License (MIT). Please see [License File](LICENSE.md) for more information.
[ico-version]: https://img.shields.io/packagist/v/teamtnt/tntsearch.svg?style=flat-square
[ico-license]: https://img.shields.io/badge/license-MIT-brightgreen.svg?style=flat-square
[ico-downloads]: https://img.shields.io/packagist/dt/teamtnt/tntsearch.svg?style=flat-square
[link-packagist]: https://packagist.org/packages/teamtnt/tntsearch
[link-downloads]: https://packagist.org/packages/teamtnt/tntsearch
[link-author]: https://github.com/nticaric
[link-contributors]: ../../contributors
---
From Croatia with ♥ by TNT Studio ([@tntstudiohr](https://twitter.com/tntstudiohr), [blog](https://tnt.studio))
@@ -0,0 +1,42 @@
{
"name": "teamtnt/tntsearch",
"type": "library",
"description": "A fully featured full text search engine written in PHP",
"keywords": [
"teamtnt",
"tntsearch",
"search",
"fulltext",
"geosearch",
"text classification",
"bm25",
"stemming",
"fuzzy search"
],
"homepage": "https://github.com/teamtnt/tntsearch",
"license": "MIT",
"authors": [{
"name": "Nenad Tičarić",
"email": "nticaric@gmail.com",
"homepage": "http://www.tntstudio.us",
"role": "Developer"
}],
"require": {
"php": "~7.1|^8",
"ext-pdo_sqlite": "*",
"ext-sqlite3": "*",
"ext-mbstring": "*"
},
"require-dev": {
"phpunit/phpunit": "7.*|8.*|9.*",
"symfony/var-dumper": "^4|^5.2"
},
"autoload": {
"psr-4": {
"TeamTNT\\TNTSearch\\": "src"
},
"files": [
"helper/helpers.php"
]
}
}
@@ -0,0 +1,25 @@
<?php
if (!function_exists('stringEndsWith')) {
function stringEndsWith($haystack, $needle)
{
// search forward starting from end minus needle length characters
return $needle === "" || (($temp = strlen($haystack) - strlen($needle)) >= 0 && strpos($haystack, $needle, $temp) !== false);
}
}
if (!function_exists('fuzzyMatch')) {
function fuzzyMatch($pattern, $items)
{
$fm = new TeamTNT\TNTSearch\TNTFuzzyMatch;
return $fm->fuzzyMatch($pattern, $items);
}
}
if (!function_exists('fuzzyMatchFromFile')) {
function fuzzyMatchFromFile($pattern, $path)
{
$fm = new TeamTNT\TNTSearch\TNTFuzzyMatch;
return $fm->fuzzyMatchFromFile($pattern, $path);
}
}
+15
View File
@@ -0,0 +1,15 @@
<?php
/*
|--------------------------------------------------------------------------
| Register The Composer Auto Loader
|--------------------------------------------------------------------------
|
| Composer provides a convenient, automatically generated class loader
| for our application. We just need to utilize it! We'll require it
| into the script here so that we do not have to worry about the
| loading of any our classes "manually". Feels great to relax.
|
*/
require __DIR__ . '/vendor/autoload.php';
+29
View File
@@ -0,0 +1,29 @@
<?xml version="1.0" encoding="UTF-8"?>
<phpunit backupGlobals="false"
backupStaticAttributes="false"
bootstrap="phpunit.php"
colors="true"
convertErrorsToExceptions="true"
convertNoticesToExceptions="true"
convertWarningsToExceptions="true"
processIsolation="false"
stopOnFailure="true"
>
<testsuites>
<testsuite name="TNTSearch Test Suite">
<directory>./tests/</directory>
</testsuite>
</testsuites>
<!-- <filter>
<whitelist>
<directory suffix=".php">./src/</directory>
</whitelist>
</filter>
<logging>
<log type="coverage-html" target="./coverage" charset="UTF-8"
yui="true" highlight="true"
lowUpperBound="50" highLowerBound="80"/>
<log type="testdox-html" target="./coverage/testdox.html" />
</logging> -->
</phpunit>
@@ -0,0 +1,131 @@
<?php
namespace TeamTNT\TNTSearch\Classifier;
use TeamTNT\TNTSearch\Stemmer\NoStemmer;
use TeamTNT\TNTSearch\Support\Tokenizer;
class TNTClassifier
{
public $documents = [];
public $words = [];
public $types = [];
public $tokenizer = null;
public $stemmer = null;
protected $arraySumOfWordType = null;
protected $arraySumOfDocuments = null;
public function __construct()
{
$this->tokenizer = new Tokenizer;
$this->stemmer = new NoStemmer;
}
public function predict($statement)
{
$words = $this->tokenizer->tokenize($statement);
$best_likelihood = -INF;
$best_type = '';
foreach ($this->types as $type) {
$likelihood = log($this->pTotal($type)); // calculate P(Type)
$p = 0;
foreach ($words as $word) {
$word = $this->stemmer->stem($word);
$p += log($this->p($word, $type));
}
$likelihood += $p; // calculate P(word, Type)
if ($likelihood > $best_likelihood) {
$best_likelihood = $likelihood;
$best_type = $type;
}
}
return [
'likelihood' => $best_likelihood,
'label' => $best_type
];
}
public function learn($statement, $type)
{
if (!in_array($type, $this->types)) {
$this->types[] = $type;
}
$words = $this->tokenizer->tokenize($statement);
foreach ($words as $word) {
$word = $this->stemmer->stem($word);
if (!isset($this->words[$type][$word])) {
$this->words[$type][$word] = 0;
}
$this->words[$type][$word]++; // increment the word count for the type
}
if (!isset($this->documents[$type])) {
$this->documents[$type] = 0;
}
$this->documents[$type]++; // increment the document count for the type
}
public function p($word, $type)
{
$count = 0;
if (isset($this->words[$type][$word])) {
$count = $this->words[$type][$word];
}
if (!isset($this->arraySumOfWordType[$type])) {
$this->arraySumOfWordType[$type] = array_sum($this->words[$type]);
}
return ($count + 1) / ($this->arraySumOfWordType[$type] + $this->vocabularyCount());
}
public function pTotal($type)
{
if (!isset($this->arraySumOfDocuments)) {
$this->arraySumOfDocuments = array_sum($this->documents);
}
return ($this->documents[$type]) / $this->arraySumOfDocuments;
}
public function vocabularyCount()
{
if (isset($this->vc)) {
return $this->vc;
}
$words = [];
foreach ($this->words as $key => $value) {
foreach ($this->words[$key] as $word => $count) {
$words[$word] = 0;
}
}
$this->vc = count($words);
return $this->vc;
}
public function save($path)
{
$s = serialize($this);
return file_put_contents($path, $s);
}
public function load($name)
{
$s = file_get_contents($name);
$classifier = unserialize($s);
unset($this->vc);
unset($this->arraySumOfDocuments);
unset($this->arraySumOfWordType);
$this->documents = $classifier->documents;
$this->words = $classifier->words;
$this->types = $classifier->types;
$this->tokenizer = $classifier->tokenizer;
$this->stemmer = $classifier->stemmer;
}
}
@@ -0,0 +1,77 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use PDO;
class Connector
{
/**
* The default PDO connection options.
*
* @var array
*/
protected $options = [
PDO::ATTR_CASE => PDO::CASE_NATURAL,
PDO::ATTR_ERRMODE => PDO::ERRMODE_EXCEPTION,
PDO::ATTR_ORACLE_NULLS => PDO::NULL_NATURAL,
PDO::ATTR_STRINGIFY_FETCHES => false,
PDO::ATTR_EMULATE_PREPARES => false,
];
/**
* Get the PDO options based on the configuration.
*
* @param array $config
* @return array
*/
public function getOptions(array $config)
{
return $this->options;
}
/**
* Create a new PDO connection.
*
* @param string $dsn
* @param array $config
* @param array $options
* @return \PDO
*/
public function createConnection($dsn, array $config, array $options)
{
extract($config, EXTR_SKIP);
if (!array_key_exists('username', $config)) {
$username = null;
}
if (!array_key_exists('password', $config)) {
$password = null;
}
return new PDO($dsn, $username, $password, $options);
}
/**
* Get the default PDO connection options.
*
* @return array
*/
public function getDefaultOptions()
{
return $this->options;
}
/**
* Set the default PDO connection options.
*
* @param array $options
* @return void
*/
public function setDefaultOptions(array $options)
{
$this->options = $options;
}
}
@@ -0,0 +1,14 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
interface ConnectorInterface
{
/**
* Establish a database connection.
*
* @param array $config
* @return \PDO
*/
public function connect(array $config);
}
@@ -0,0 +1,21 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use Exception;
class FileSystemConnector extends Connector implements ConnectorInterface
{
/**
* Establish a database connection.
*
* @param array $config
* @return \PDO
*
* @throws \InvalidArgumentException
*/
public function connect(array $config)
{
}
}
@@ -0,0 +1,139 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use PDO;
class MySqlConnector extends Connector implements ConnectorInterface
{
/**
* Establish a database connection.
*
* @param array $config
* @return \PDO
*/
public function connect(array $config)
{
$dsn = $this->getDsn($config);
$options = $this->getOptions($config);
// We need to grab the PDO options that should be used while making the brand
// new connection instance. The PDO options control various aspects of the
// connection's behavior, and some might be specified by the developers.
$connection = $this->createConnection($dsn, $config, $options);
if (! empty($config['database'])) {
$connection->exec("use `{$config['database']}`;");
}
$collation = 'utf8_unicode_ci';
if (! empty($config['collation'])) {
$collation = $config['collation'];
}
// Next we will set the "names" and "collation" on the clients connections so
// a correct character set will be used by this client. The collation also
// is set on the server but needs to be set here on this client objects.
if (isset($config['charset'])) {
$charset = $config['charset'];
$names = "set names '{$charset}'".
(! is_null($collation) ? " collate '{$collation}'" : '');
$connection->prepare($names)->execute();
}
// Next, we will check to see if a timezone has been specified in this config
// and if it has we will issue a statement to modify the timezone with the
// database. Setting this DB timezone is an optional configuration item.
if (isset($config['timezone'])) {
$connection->prepare(
'set time_zone="'.$config['timezone'].'"'
)->execute();
}
$this->setModes($connection, $config);
return $connection;
}
public function getOptions(array $config)
{
return array_merge(parent::getOptions($config), [
PDO::MYSQL_ATTR_USE_BUFFERED_QUERY => false,
]);
}
/**
* Create a DSN string from a configuration.
*
* Chooses socket or host/port based on the 'unix_socket' config value.
*
* @param array $config
* @return string
*/
protected function getDsn(array $config)
{
return $this->configHasSocket($config) ? $this->getSocketDsn($config) : $this->getHostDsn($config);
}
/**
* Determine if the given configuration array has a UNIX socket value.
*
* @param array $config
* @return bool
*/
protected function configHasSocket(array $config)
{
return isset($config['unix_socket']) && ! empty($config['unix_socket']);
}
/**
* Get the DSN string for a socket configuration.
*
* @param array $config
* @return string
*/
protected function getSocketDsn(array $config)
{
return "mysql:unix_socket={$config['unix_socket']};dbname={$config['database']}";
}
/**
* Get the DSN string for a host / port configuration.
*
* @param array $config
* @return string
*/
protected function getHostDsn(array $config)
{
extract($config, EXTR_SKIP);
return isset($port)
? "mysql:host={$host};port={$port};dbname={$database}"
: "mysql:host={$host};dbname={$database}";
}
/**
* Set the modes for the connection.
*
* @param \PDO $connection
* @param array $config
* @return void
*/
protected function setModes(PDO $connection, array $config)
{
if (isset($config['modes'])) {
$modes = implode(',', $config['modes']);
$connection->prepare("set session sql_mode='{$modes}'")->execute();
} elseif (isset($config['strict'])) {
if ($config['strict']) {
$connection->prepare("set session sql_mode='ONLY_FULL_GROUP_BY,STRICT_TRANS_TABLES,NO_ZERO_IN_DATE,NO_ZERO_DATE,ERROR_FOR_DIVISION_BY_ZERO,NO_AUTO_CREATE_USER,NO_ENGINE_SUBSTITUTION'")->execute();
} else {
$connection->prepare("set session sql_mode='NO_ENGINE_SUBSTITUTION'")->execute();
}
}
}
}
@@ -0,0 +1,121 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use PDO;
class PostgresConnector extends Connector implements ConnectorInterface
{
/**
* The default PDO connection options.
*
* @var array
*/
protected $options = [
PDO::ATTR_CASE => PDO::CASE_NATURAL,
PDO::ATTR_ERRMODE => PDO::ERRMODE_EXCEPTION,
PDO::ATTR_ORACLE_NULLS => PDO::NULL_NATURAL,
PDO::ATTR_STRINGIFY_FETCHES => false,
];
/**
* Establish a database connection.
*
* @param array $config
* @return \PDO
*/
public function connect(array $config)
{
// First we'll create the basic DSN and connection instance connecting to the
// using the configuration option specified by the developer. We will also
// set the default character set on the connections to UTF-8 by default.
$dsn = $this->getDsn($config);
$options = $this->getOptions($config);
$connection = $this->createConnection($dsn, $config, $options);
$charset = 'utf8';
if (isset($config['charset'])) {
$charset = $config['charset'];
}
$connection->prepare("set names '$charset'")->execute();
// Next, we will check to see if a timezone has been specified in this config
// and if it has we will issue a statement to modify the timezone with the
// database. Setting this DB timezone is an optional configuration item.
if (isset($config['timezone'])) {
$timezone = $config['timezone'];
$connection->prepare("set time zone '$timezone'")->execute();
}
// Unlike MySQL, Postgres allows the concept of "schema" and a default schema
// may have been specified on the connections. If that is the case we will
// set the default schema search paths to the specified database schema.
if (isset($config['schema'])) {
$schema = $this->formatSchema($config['schema']);
$connection->prepare("set search_path to {$schema}")->execute();
}
// Postgres allows an application_name to be set by the user and this name is
// used to when monitoring the application with pg_stat_activity. So we'll
// determine if the option has been specified and run a statement if so.
if (isset($config['application_name'])) {
$applicationName = $config['application_name'];
$connection->prepare("set application_name to '$applicationName'")->execute();
}
return $connection;
}
/**
* Create a DSN string from a configuration.
*
* @param array $config
* @return string
*/
protected function getDsn(array $config)
{
// First we will create the basic DSN setup as well as the port if it is in
// in the configuration options. This will give us the basic DSN we will
// need to establish the PDO connections and return them back for use.
extract($config, EXTR_SKIP);
$host = isset($host) ? "host={$host};" : '';
$dsn = "pgsql:{$host}dbname={$database}";
// If a port was specified, we will add it to this Postgres DSN connections
// format. Once we have done that we are ready to return this connection
// string back out for usage, as this has been fully constructed here.
if (isset($config['port'])) {
$dsn .= ";port={$port}";
}
if (isset($config['sslmode'])) {
$dsn .= ";sslmode={$sslmode}";
}
return $dsn;
}
/**
* Format the schema for the DSN.
*
* @param array|string $schema
* @return string
*/
protected function formatSchema($schema)
{
if (is_array($schema)) {
return '"'.implode('", "', $schema).'"';
} else {
return '"'.$schema.'"';
}
}
}
@@ -0,0 +1,40 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use Exception;
class SQLiteConnector extends Connector implements ConnectorInterface
{
protected $options = [];
/**
* Establish a database connection.
*
* @param array $config
* @return \PDO
*
* @throws \InvalidArgumentException
*/
public function connect(array $config)
{
$options = $this->getOptions($config);
// SQLite supports "in-memory" databases that only last as long as the owning
// connection does. These are useful for tests or for short lifetime store
// querying. In-memory databases may only have a single open connection.
if ($config['database'] == ':memory:') {
return $this->createConnection('sqlite::memory:', $config, $options);
}
$path = realpath($config['database']);
// Here we'll verify that the SQLite database exists before going any further
// as the developer probably wants to know if the database exists and this
// SQLite driver will not throw any exception if it does not by default.
if ($path === false) {
throw new Exception("Database (${config['database']}) does not exist.");
}
return $this->createConnection("sqlite:{$path}", $config, $options);
}
}
@@ -0,0 +1,71 @@
<?php
namespace TeamTNT\TNTSearch\Connectors;
use PDO;
class SqlServerConnector extends Connector implements ConnectorInterface {
/**
* The PDO connection options.
*
* @var array
*/
protected $options = array(
PDO::ATTR_CASE => PDO::CASE_NATURAL,
PDO::ATTR_ERRMODE => PDO::ERRMODE_EXCEPTION,
PDO::ATTR_ORACLE_NULLS => PDO::NULL_NATURAL,
PDO::ATTR_STRINGIFY_FETCHES => false,
);
/**
* Establish a database connection.
*
* @param array $config
* @return PDO
*/
public function connect(array $config)
{
$options = $this->getOptions($config);
return $this->createConnection($this->getDsn($config), $config, $options);
}
/**
* Create a DSN string from a configuration.
*
* @param array $config
* @return string
*/
protected function getDsn(array $config)
{
extract($config);
// First we will create the basic DSN setup as well as the port if it is in
// in the configuration options. This will give us the basic DSN we will
// need to establish the PDO connections and return them back for use.
$port = isset($config['port']) ? ','.$port : '';
if (in_array('dblib', $this->getAvailableDrivers()))
{
return "dblib:host={$host}{$port};dbname={$database}";
}
else
{
$dbName = $database != '' ? ";Database={$database}" : '';
return "sqlsrv:Server={$host}{$port}{$dbName}";
}
}
/**
* Get the available PDO drivers.
*
* @return array
*/
protected function getAvailableDrivers()
{
return PDO::getAvailableDrivers();
}
}
@@ -0,0 +1,9 @@
<?php
namespace TeamTNT\TNTSearch\Exceptions;
use Exception;
class IndexNotFoundException extends Exception
{
}
@@ -0,0 +1,16 @@
<?php
namespace TeamTNT\TNTSearch\FileReaders;
use SplFileInfo;
interface FileReaderInterface
{
/**
* Read the content of a file
*
* @param SplFileInfo $fileinfo
* @return string
*/
public function read(SplFileInfo $fileinfo);
}
@@ -0,0 +1,16 @@
<?php
namespace TeamTNT\TNTSearch\FileReaders;
use SplFileInfo;
class TextFileReader implements FileReaderInterface
{
public $fileMapCallback = null;
public $fileFilterCallback = null;
public function read(SplFileInfo $fileinfo)
{
return file_get_contents($fileinfo);
}
}
@@ -0,0 +1,72 @@
<?php
namespace TeamTNT\TNTSearch\Indexer;
use PDO;
class TNTGeoIndexer extends TNTIndexer
{
public function createIndex($indexName)
{
$this->indexName = $indexName;
if (file_exists($this->config['storage'].$indexName)) {
unlink($this->config['storage'].$indexName);
}
$this->index = new PDO('sqlite:'.$this->config['storage'].$indexName);
$this->index->setAttribute(PDO::ATTR_ERRMODE, PDO::ERRMODE_EXCEPTION);
$this->index->exec("CREATE TABLE IF NOT EXISTS locations (
doc_id INTEGER,
longitude REAL,
latitude REAL,
cos_lat REAL,
sin_lat REAL,
cos_lng REAL,
sin_lng REAL
)");
$this->index->exec("CREATE INDEX location_index ON locations ('longitude', 'latitude');");
$this->index->exec("CREATE TABLE IF NOT EXISTS info (key TEXT, value INTEGER)");
$connector = $this->createConnector($this->config);
if (!$this->dbh) {
$this->dbh = $connector->connect($this->config);
}
return $this;
}
public function processDocument($row)
{
$this->prepareInsertStatement();
$docId = $row->get($this->getPrimaryKey());
$longitude = $row->get('longitude');
$latitude = $row->get('latitude');
$cos_lat = cos($latitude * pi() / 180);
$sin_lat = sin($latitude * pi() / 180);
$cos_lng = cos($longitude * pi() / 180);
$sin_lng = sin($longitude * pi() / 180);
$this->insertStmt->bindParam(":doc_id", $docId);
$this->insertStmt->bindParam(":longitude", $longitude);
$this->insertStmt->bindParam(":latitude", $latitude);
$this->insertStmt->bindParam(":cos_lat", $cos_lat);
$this->insertStmt->bindParam(":sin_lat", $sin_lat);
$this->insertStmt->bindParam(":cos_lng", $cos_lng);
$this->insertStmt->bindParam(":sin_lng", $sin_lng);
$this->insertStmt->execute();
}
public function prepareInsertStatement()
{
if (isset($this->insertStmt)) {
return $this->insertStmt;
}
$this->insertStmt = $this->index->prepare("INSERT INTO locations (doc_id, longitude, latitude, cos_lat, sin_lat, cos_lng, sin_lng)
VALUES (:doc_id, :longitude, :latitude, :cos_lat, :sin_lat, :cos_lng, :sin_lng)");
}
}
@@ -0,0 +1,695 @@
<?php
namespace TeamTNT\TNTSearch\Indexer;
use Exception;
use PDO;
use RecursiveDirectoryIterator;
use RecursiveIteratorIterator;
use TeamTNT\TNTSearch\Connectors\FileSystemConnector;
use TeamTNT\TNTSearch\Connectors\MySqlConnector;
use TeamTNT\TNTSearch\Connectors\PostgresConnector;
use TeamTNT\TNTSearch\Connectors\SQLiteConnector;
use TeamTNT\TNTSearch\Connectors\SqlServerConnector;
use TeamTNT\TNTSearch\FileReaders\TextFileReader;
use TeamTNT\TNTSearch\Stemmer\CroatianStemmer;
use TeamTNT\TNTSearch\Stemmer\NoStemmer;
use TeamTNT\TNTSearch\Support\Collection;
use TeamTNT\TNTSearch\Support\Tokenizer;
use TeamTNT\TNTSearch\Support\TokenizerInterface;
class TNTIndexer
{
protected $index = null;
protected $dbh = null;
protected $primaryKey = null;
protected $excludePrimaryKey = true;
public $stemmer = null;
public $tokenizer = null;
public $stopWords = [];
public $filereader = null;
public $config = [];
protected $query = "";
protected $wordlist = [];
protected $inMemoryTerms = [];
protected $decodeHTMLEntities = false;
public $disableOutput = false;
public $inMemory = true;
public $steps = 1000;
public $indexName = "";
public $statementsPrepared = false;
public function __construct()
{
$this->stemmer = new NoStemmer;
$this->tokenizer = new Tokenizer;
$this->filereader = new TextFileReader;
}
/**
* @param TokenizerInterface $tokenizer
*/
public function setTokenizer(TokenizerInterface $tokenizer)
{
$this->tokenizer = $tokenizer;
$this->updateInfoTable('tokenizer', get_class($tokenizer));
}
public function setStopWords(array $stopWords)
{
$this->stopWords = $stopWords;
}
/**
* @param array $config
*/
public function loadConfig(array $config)
{
$this->config = $config;
$this->config['storage'] = rtrim($this->config['storage'], '/').'/';
if (!isset($this->config['driver'])) {
$this->config['driver'] = "";
}
if (!isset($this->config['wal'])) {
$this->config['wal'] = true;
}
}
/**
* @return string
*/
public function getStoragePath()
{
return $this->config['storage'];
}
public function getStemmer()
{
return $this->stemmer;
}
/**
* @return string
*/
public function getPrimaryKey()
{
if (isset($this->primaryKey)) {
return $this->primaryKey;
}
return 'id';
}
/**
* @param string $primaryKey
*/
public function setPrimaryKey($primaryKey)
{
$this->primaryKey = $primaryKey;
}
public function excludePrimaryKey()
{
$this->excludePrimaryKey = true;
}
public function includePrimaryKey()
{
$this->excludePrimaryKey = false;
}
public function setStemmer($stemmer)
{
$this->stemmer = $stemmer;
$this->updateInfoTable('stemmer', get_class($stemmer));
}
public function setCroatianStemmer()
{
$this->setStemmer(new CroatianStemmer);
}
/**
* @param string $language - one of: no, arabic, croatian, german, italian, porter, portuguese, russian, ukrainian
*/
public function setLanguage($language = 'no')
{
$class = 'TeamTNT\\TNTSearch\\Stemmer\\'.ucfirst(strtolower($language)).'Stemmer';
$this->setStemmer(new $class);
}
/**
* @param PDO $index
*/
public function setIndex($index)
{
$this->index = $index;
}
public function setFileReader($filereader)
{
$this->filereader = $filereader;
}
public function prepareStatementsForIndex()
{
if (!$this->statementsPrepared) {
$this->insertWordlistStmt = $this->index->prepare("INSERT INTO wordlist (term, num_hits, num_docs) VALUES (:keyword, :hits, :docs)");
$this->selectWordlistStmt = $this->index->prepare("SELECT * FROM wordlist WHERE term like :keyword LIMIT 1");
$this->updateWordlistStmt = $this->index->prepare("UPDATE wordlist SET num_docs = num_docs + :docs, num_hits = num_hits + :hits WHERE term = :keyword");
$this->statementsPrepared = true;
}
}
/**
* @param string $indexName
*
* @return TNTIndexer
*/
public function createIndex($indexName)
{
$this->indexName = $indexName;
if (file_exists($this->config['storage'].$indexName)) {
unlink($this->config['storage'].$indexName);
}
$this->index = new PDO('sqlite:'.$this->config['storage'].$indexName);
$this->index->setAttribute(PDO::ATTR_ERRMODE, PDO::ERRMODE_EXCEPTION);
if ($this->config['wal']) {
$this->index->exec("PRAGMA journal_mode=wal;");
}
$this->index->exec("CREATE TABLE IF NOT EXISTS wordlist (
id INTEGER PRIMARY KEY,
term TEXT UNIQUE COLLATE nocase,
num_hits INTEGER,
num_docs INTEGER)");
$this->index->exec("CREATE UNIQUE INDEX 'main'.'index' ON wordlist ('term');");
$this->index->exec("CREATE TABLE IF NOT EXISTS doclist (
term_id INTEGER,
doc_id INTEGER,
hit_count INTEGER)");
$this->index->exec("CREATE TABLE IF NOT EXISTS fields (
id INTEGER PRIMARY KEY,
name TEXT)");
$this->index->exec("CREATE TABLE IF NOT EXISTS hitlist (
term_id INTEGER,
doc_id INTEGER,
field_id INTEGER,
position INTEGER,
hit_count INTEGER)");
$this->index->exec("CREATE TABLE IF NOT EXISTS info (
key TEXT,
value INTEGER)");
$this->index->exec("INSERT INTO info ( 'key', 'value') values ( 'total_documents', 0)");
$this->index->exec("INSERT INTO info ( 'key', 'value') values ( 'stemmer', 'TeamTNT\TNTSearch\Stemmer\NoStemmer')");
$this->index->exec("INSERT INTO info ( 'key', 'value') values ( 'tokenizer', 'TeamTNT\TNTSearch\Support\Tokenizer')");
$this->index->exec("CREATE INDEX IF NOT EXISTS 'main'.'term_id_index' ON doclist ('term_id' COLLATE BINARY);");
$this->index->exec("CREATE INDEX IF NOT EXISTS 'main'.'doc_id_index' ON doclist ('doc_id');");
if (isset($this->config['stemmer'])) {
$this->setStemmer(new $this->config['stemmer']);
}
if (isset($this->config['tokenizer'])) {
$this->setTokenizer(new $this->config['tokenizer']);
}
if (!$this->dbh) {
$connector = $this->createConnector($this->config);
$this->dbh = $connector->connect($this->config);
}
return $this;
}
public function indexBeginTransaction()
{
$this->index->beginTransaction();
}
public function indexEndTransaction()
{
$this->index->commit();
}
/**
* @param array $config
*
* @return FileSystemConnector|MySqlConnector|PostgresConnector|SQLiteConnector|SqlServerConnector
* @throws Exception
*/
public function createConnector(array $config)
{
if (!isset($config['driver'])) {
throw new Exception('A driver must be specified.');
}
switch ($config['driver']) {
case 'mysql':
return new MySqlConnector;
case 'pgsql':
return new PostgresConnector;
case 'sqlite':
return new SQLiteConnector;
case 'sqlsrv':
return new SqlServerConnector;
case 'filesystem':
return new FileSystemConnector;
}
throw new Exception("Unsupported driver [{$config['driver']}]");
}
/**
* @param PDO $dbh
*/
public function setDatabaseHandle(PDO $dbh)
{
$this->dbh = $dbh;
if ($this->dbh->getAttribute(PDO::ATTR_DRIVER_NAME) == 'mysql') {
$this->dbh->setAttribute(PDO::MYSQL_ATTR_USE_BUFFERED_QUERY, false);
}
}
public function query($query)
{
$this->query = $query;
}
public function run()
{
if ($this->config['driver'] == "filesystem") {
return $this->readDocumentsFromFileSystem();
}
$result = $this->dbh->query($this->query);
$counter = 0;
$this->index->beginTransaction();
while ($row = $result->fetch(PDO::FETCH_ASSOC)) {
$counter++;
$this->processDocument(new Collection($row));
if ($counter % $this->steps == 0) {
$this->info("Processed $counter rows");
}
if ($counter % 10000 == 0) {
$this->index->commit();
$this->index->beginTransaction();
$this->info("Committed");
}
}
$this->index->commit();
$this->updateInfoTable('total_documents', $counter);
$this->info("Total rows $counter");
}
public function readDocumentsFromFileSystem()
{
$exclude = [];
if (isset($this->config['exclude'])) {
$exclude = $this->config['exclude'];
}
$this->index->exec("CREATE TABLE IF NOT EXISTS filemap (
id INTEGER PRIMARY KEY,
path TEXT)");
$path = realpath($this->config['location']);
$objects = new RecursiveIteratorIterator(new RecursiveDirectoryIterator($path), RecursiveIteratorIterator::SELF_FIRST);
$this->index->beginTransaction();
$counter = 0;
foreach ($objects as $name => $object) {
$name = str_replace($path.'/', '', $name);
if (is_callable($this->config['extension'])) {
$includeFile = $this->config['extension']($object);
} elseif (is_array($this->config['extension'])) {
$includeFile = in_array($object->getExtension(), $this->config['extension']);
} else {
$includeFile = stringEndsWith($name, $this->config['extension']);
}
if ($includeFile && !in_array($name, $exclude)) {
$counter++;
$file = [
'id' => $counter,
'name' => $name,
'content' => $this->filereader->read($object)
];
$fileCollection = new Collection($file);
if (property_exists($this->filereader, 'fileFilterCallback')
&& is_callable($this->filereader->fileFilterCallback)) {
$fileCollection = $fileCollection->filter($this->filereader->fileFilterCallback);
}
if (property_exists($this->filereader, 'fileMapCallback')
&& is_callable($this->filereader->fileMapCallback)) {
$fileCollection = $fileCollection->map($this->filereader->fileMapCallback);
}
$this->processDocument($fileCollection);
$statement = $this->index->prepare("INSERT INTO filemap ( 'id', 'path') values ( $counter, :object)");
$statement->bindParam(':object', $object);
$statement->execute();
$this->info("Processed $counter $object");
}
}
$this->index->commit();
$this->index->exec("INSERT INTO info ( 'key', 'value') values ( 'total_documents', $counter)");
$this->index->exec("INSERT INTO info ( 'key', 'value') values ( 'driver', 'filesystem')");
$this->info("Total rows $counter");
$this->info("Index created: {$this->config['storage']}");
}
public function processDocument($row)
{
$documentId = $row->get($this->getPrimaryKey());
if ($this->excludePrimaryKey) {
$row->forget($this->getPrimaryKey());
}
$stems = $row->map(function ($columnContent, $columnName) use ($row) {
return $this->stemText($columnContent);
});
$this->saveToIndex($stems, $documentId);
}
public function insert($document)
{
$this->processDocument(new Collection($document));
$total = $this->totalDocumentsInCollection() + 1;
$this->updateInfoTable('total_documents', $total);
}
public function update($id, $document)
{
$this->delete($id);
$this->insert($document);
}
public function delete($documentId)
{
$rows = $this->prepareAndExecuteStatement("SELECT * FROM doclist WHERE doc_id = :documentId;", [
['key' => ':documentId', 'value' => $documentId]
])->fetchAll(PDO::FETCH_ASSOC);
$updateStmt = $this->index->prepare("UPDATE wordlist SET num_docs = num_docs - 1, num_hits = num_hits - :hits WHERE id = :term_id");
foreach ($rows as $document) {
$updateStmt->bindParam(":hits", $document['hit_count']);
$updateStmt->bindParam(":term_id", $document['term_id']);
$updateStmt->execute();
}
$this->prepareAndExecuteStatement("DELETE FROM doclist WHERE doc_id = :documentId;", [
['key' => ':documentId', 'value' => $documentId]
]);
$res = $this->prepareAndExecuteStatement("DELETE FROM wordlist WHERE num_hits = 0");
$affected = $res->rowCount();
if ($affected) {
$total = $this->totalDocumentsInCollection() - 1;
$this->updateInfoTable('total_documents', $total);
}
}
public function updateInfoTable($key, $value)
{
$this->updateInfoTableStmt = $this->index->prepare("UPDATE info SET value = :value WHERE key = :key");
$this->updateInfoTableStmt->bindValue(':key', $key);
$this->updateInfoTableStmt->bindValue(':value', $value);
$this->updateInfoTableStmt->execute();
}
public function stemText($text)
{
$stemmer = $this->getStemmer();
$words = $this->breakIntoTokens($text);
$stems = [];
foreach ($words as $word) {
$stems[] = $stemmer->stem($word);
}
return $stems;
}
public function breakIntoTokens($text)
{
if ($this->decodeHTMLEntities) {
$text = html_entity_decode($text);
}
return $this->tokenizer->tokenize($text, $this->stopWords);
}
public function decodeHtmlEntities($value = true)
{
$this->decodeHTMLEntities = $value;
}
public function saveToIndex($stems, $docId)
{
$this->prepareStatementsForIndex();
$terms = $this->saveWordlist($stems);
$this->saveDoclist($terms, $docId);
$this->saveHitList($stems, $docId, $terms);
}
/**
* @param $stems
*
* @return array
*/
public function saveWordlist($stems)
{
$terms = [];
$stems->map(function ($column, $key) use (&$terms) {
foreach ($column as $term) {
if (array_key_exists($term, $terms)) {
$terms[$term]['hits']++;
$terms[$term]['docs'] = 1;
} else {
$terms[$term] = [
'hits' => 1,
'docs' => 1,
'id' => 0
];
}
}
});
foreach ($terms as $key => $term) {
try {
$this->insertWordlistStmt->bindParam(":keyword", $key);
$this->insertWordlistStmt->bindParam(":hits", $term['hits']);
$this->insertWordlistStmt->bindParam(":docs", $term['docs']);
$this->insertWordlistStmt->execute();
$terms[$key]['id'] = $this->index->lastInsertId();
if ($this->inMemory) {
$this->inMemoryTerms[$key] = $terms[$key]['id'];
}
} catch (\Exception $e) {
if ($e->getCode() == 23000) {
$this->updateWordlistStmt->bindValue(':docs', $term['docs']);
$this->updateWordlistStmt->bindValue(':hits', $term['hits']);
$this->updateWordlistStmt->bindValue(':keyword', $key);
$this->updateWordlistStmt->execute();
if (!$this->inMemory) {
$this->selectWordlistStmt->bindValue(':keyword', $key);
$this->selectWordlistStmt->execute();
$res = $this->selectWordlistStmt->fetch(PDO::FETCH_ASSOC);
$terms[$key]['id'] = $res['id'];
} else {
$terms[$key]['id'] = $this->inMemoryTerms[$key];
}
} else {
echo "Error while saving wordlist: ".$e->getMessage()."\n";
}
// Statements must be refreshed, because in this state they have error attached to them.
$this->statementsPrepared = false;
$this->prepareStatementsForIndex();
}
}
return $terms;
}
public function saveDoclist($terms, $docId)
{
$insert = "INSERT INTO doclist (term_id, doc_id, hit_count) VALUES (:id, :doc, :hits)";
$stmt = $this->index->prepare($insert);
foreach ($terms as $key => $term) {
$stmt->bindValue(':id', $term['id']);
$stmt->bindValue(':doc', $docId);
$stmt->bindValue(':hits', $term['hits']);
try {
$stmt->execute();
} catch (\Exception $e) {
//we have a duplicate
echo $e->getMessage();
}
}
}
public function saveHitList($stems, $docId, $termsList)
{
return;
$fieldCounter = 0;
$fields = [];
$insert = "INSERT INTO hitlist (term_id, doc_id, field_id, position, hit_count)
VALUES (:term_id, :doc_id, :field_id, :position, :hit_count)";
$stmt = $this->index->prepare($insert);
foreach ($stems as $field => $terms) {
$fields[$fieldCounter] = $field;
$positionCounter = 0;
$termCounts = array_count_values($terms);
foreach ($terms as $term) {
if (isset($termsList[$term])) {
$stmt->bindValue(':term_id', $termsList[$term]['id']);
$stmt->bindValue(':doc_id', $docId);
$stmt->bindValue(':field_id', $fieldCounter);
$stmt->bindValue(':position', $positionCounter);
$stmt->bindValue(':hit_count', $termCounts[$term]);
$stmt->execute();
}
$positionCounter++;
}
$fieldCounter++;
}
}
public function getWordFromWordList($word)
{
$selectStmt = $this->index->prepare("SELECT * FROM wordlist WHERE term like :keyword LIMIT 1");
$selectStmt->bindValue(':keyword', $word);
$selectStmt->execute();
return $selectStmt->fetch(PDO::FETCH_ASSOC);
}
/**
* @param $word
*
* @return int
*/
public function countWordInWordList($word)
{
$res = $this->getWordFromWordList($word);
if ($res) {
return $res['num_hits'];
}
return 0;
}
/**
* @param $word
*
* @return int
*/
public function countDocHitsInWordList($word)
{
$res = $this->getWordFromWordList($word);
if ($res) {
return $res['num_docs'];
}
return 0;
}
public function buildDictionary($filename, $count = -1, $hits = true, $docs = false)
{
$selectStmt = $this->index->prepare("SELECT * FROM wordlist ORDER BY num_hits DESC;");
$selectStmt->execute();
$dictionary = "";
$counter = 0;
while ($row = $selectStmt->fetch(PDO::FETCH_ASSOC)) {
$dictionary .= $row['term'];
if ($hits) {
$dictionary .= "\t".$row['num_hits'];
}
if ($docs) {
$dictionary .= "\t".$row['num_docs'];
}
$counter++;
if ($counter >= $count && $count > 0) {
break;
}
$dictionary .= "\n";
}
file_put_contents($filename, $dictionary, LOCK_EX);
}
/**
* @return int
*/
public function totalDocumentsInCollection()
{
$query = "SELECT * FROM info WHERE key = 'total_documents'";
$docs = $this->index->query($query);
return $docs->fetch(PDO::FETCH_ASSOC)['value'];
}
/**
* @param $keyword
*
* @return string
*/
public function buildTrigrams($keyword)
{
$t = "__".$keyword."__";
$trigrams = "";
for ($i = 0; $i < strlen($t) - 2; $i++) {
$trigrams .= mb_substr($t, $i, 3)." ";
}
return trim($trigrams);
}
public function prepareAndExecuteStatement($query, $params = [])
{
$statemnt = $this->index->prepare($query);
foreach ($params as $param) {
$statemnt->bindParam($param['key'], $param['value']);
}
$statemnt->execute();
return $statemnt;
}
public function info($text)
{
if (!$this->disableOutput) {
echo $text.PHP_EOL;
}
}
}
@@ -0,0 +1,147 @@
<?php
namespace TeamTNT\TNTSearch\KeywordExtraction;
class Rake
{
public function __construct($language = "english")
{
$stopwords = file_get_contents(__DIR__."/../Stopwords/".$language.".json");
$this->stopwords = json_decode($stopwords);
}
public function extractKeywords($text, $includeScores = true)
{
$phraseList = $this->generateCandidateKeywords($text);
$wordScores = $this->calculateWordScores($phraseList);
$phraseScores = $this->calculatePhraseScores($phraseList, $wordScores);
arsort($phraseScores);
$oneThird = ceil(count($phraseScores) / 3) + 1;
$phraseScores = array_slice($phraseScores, 0, $oneThird);
if ($includeScores) {
return $phraseScores;
}
return array_keys($phraseScores);
}
public function generateCandidateKeywords($text)
{
$phraseList = [];
$words = $this->tokenize($text);
$phrase = [];
foreach ($words as $word) {
if (in_array($word, $this->stopwords) || ctype_punct($word)) {
if (count($phrase) > 0) {
$phraseList[] = $phrase;
$phrase = [];
}
} else {
$phrase[] = $word;
}
}
if (count($phrase) > 0) {
$phraseList[] = $phrase;
$phrase = [];
}
return $phraseList;
}
public function calculatePhraseScores($phraseList, $wordScores)
{
$result = [];
foreach ($phraseList as $phrase) {
$wordScore = 0;
foreach ($phrase as $word) {
$wordScore += $wordScores[$word];
}
$result[implode(" ", $phrase)] = $wordScore;
}
return $result;
}
public function calculateWordScores($phraseList)
{
$result = [];
foreach ($phraseList as $phrase) {
foreach ($phrase as $word) {
$wordScore = $this->wordDegree($word, $phraseList) / $this->wordFrequency($word, $phraseList);
$result[$word] = $wordScore;
}
}
return $result;
}
public function wordDegree($word, $phraseList)
{
$count = 0;
foreach ($phraseList as $phrase) {
foreach ($phrase as $p) {
if ($p == $word) {
$count += count($phrase);
}
}
}
return $count;
}
public function wordFrequency($word, $phraseList)
{
$count = 0;
foreach ($phraseList as $phrase) {
foreach ($phrase as $p) {
if ($p == $word) {
$count++;
}
}
}
return $count;
}
public function returnFormatedPharaseList($phraseList)
{
$formatedList = [];
foreach ($phraseList as $phrase) {
$formatedList[] = implode(" ", $phrase);
}
return $formatedList;
}
public function tokenize($str)
{
$str = mb_strtolower($str);
$arr = [];
// for the character classes
// see http://php.net/manual/en/regexp.reference.unicode.php
$pat = '/
([\pZ\pC]*) # match any separator or other
# in sequence
(
[^\pP\pZ\pC]+ | # match a sequence of characters
# that are not punctuation,
# separator or other
. # match punctuations one by one
)
([\pZ\pC]*) # match a sequence of separators
# that follows
/xu';
preg_match_all($pat, $str, $arr);
return $arr[2];
}
}
@@ -0,0 +1,114 @@
<?php
namespace TeamTNT\TNTSearch\Spell;
class JaroWinklerDistance
{
private $threshold = 0.7;
public function getDistance($str1, $str2)
{
$j = $this->jaro($str1, $str2);
if ($j < $this->threshold) {
return $j;
}
$lengthOfCommonPrefix = 0;
for ($i = 0; $i < min(strlen($str1), strlen($str2)); $i++) {
if ($str1[$i] == $str2[$i]) {
$lengthOfCommonPrefix++;
} else {
break;
}
}
$lp = min(0.1, 1 / max(strlen($str1), strlen($str2))) * $lengthOfCommonPrefix;
$jw = $j + ($lp * (1 - $j));
return $jw;
}
public function jaro($str1, $str2)
{
// length of the strings
$str1_len = strlen($str1);
$str2_len = strlen($str2);
// if both strings are empty return 1
// if only one of the strings is empty return 0
if ($str1_len == 0) {
return $str2_len == 0 ? 1 : 0;
}
// max distance between two chars to be considered matching
$match_distance = max($str1_len, $str2_len) / 2 - 1;
$str1_matches = array_fill(0, $str1_len, 0);
$str2_matches = array_fill(0, $str2_len, 0);
// number of matches and transpositions
$matches = 0;
$transpositions = 0;
// find the matches
for ($i = 0; $i < $str1_len; $i++) {
// start and end take into account the match distance
$start = (int) max(0, $i - $match_distance);
$end = (int) min($i + $match_distance + 1, $str2_len);
for ($k = $start; $k < $end; $k++) {
// if $str2 already has a match continue
if ($str2_matches[$k]) {
continue;
}
// if str1 and str2 are not
if ($str1[$i] != $str2[$k]) {
continue;
}
// otherwise assume there is a match
$str1_matches[$i] = true;
$str2_matches[$k] = true;
$matches++;
break;
}
}
// if there are no matches return 0
if ($matches == 0) {
return 0.0;
}
// count transpositions
$k = 0;
for ($i = 0; $i < $str1_len; $i++) {
// if there are no matches in str1 continue
if (!$str1_matches[$i]) {
continue;
}
// while there is no match in str2 increment k
while (!$str2_matches[$k]) {
$k++;
}
// increment transpositions
if ($str1[$i] != $str2[$k]) {
$transpositions++;
}
$k++;
}
// divide the number of transpositions by two as per the algorithm specs
// this division is valid because the counted transpositions include both
// instances of the transposed characters.
$transpositions /= 2.0;
// return the Jaro distance
return (($matches / $str1_len) +
($matches / $str2_len) +
(($matches - $transpositions) / $matches)) / 3.0;
}
}
@@ -0,0 +1,129 @@
<?php
/**
* This is a reimplementation of AR-PHP Arabic stemmer.
* The original author is Khaled Al-Sham'aa <khaled@ar-php.org>
*/
namespace TeamTNT\TNTSearch\Stemmer;
class ArabicStemmer implements Stemmer
{
private static $_verbPre = 'وأسفلي';
private static $_verbPost = 'ومكانيه';
private static $_verbMay;
private static $_verbMaxPre = 4;
private static $_verbMaxPost = 6;
private static $_verbMinStem = 2;
private static $_nounPre = 'ابفكلوأ';
private static $_nounPost = 'اتةكمنهوي';
private static $_nounMay;
private static $_nounMaxPre = 4;
private static $_nounMaxPost = 6;
private static $_nounMinStem = 2;
/**
* Loads initialize values
*
* @ignore
*/
public function __construct()
{
self::$_verbMay = self::$_verbPre . self::$_verbPost;
self::$_nounMay = self::$_nounPre . self::$_nounPost;
}
/**
* Get rough stem of the given Arabic word
*
* @param string $word Arabic word you would like to get its stem
*
* @return string Arabic stem of the word
* @author Khaled Al-Sham'aa <khaled@ar-php.org>
*/
public static function stem($word)
{
$nounStem = self::roughStem(
$word, self::$_nounMay, self::$_nounPre, self::$_nounPost,
self::$_nounMaxPre, self::$_nounMaxPost, self::$_nounMinStem
);
$verbStem = self::roughStem(
$word, self::$_verbMay, self::$_verbPre, self::$_verbPost,
self::$_verbMaxPre, self::$_verbMaxPost, self::$_verbMinStem
);
if (mb_strlen($nounStem, 'UTF-8') < mb_strlen($verbStem, 'UTF-8')) {
$stem = $nounStem;
} else {
$stem = $verbStem;
}
return $stem;
}
/**
* Get rough stem of the given Arabic word (under specific rules)
*
* @param string $word Arabic word you would like to get its stem
* @param string $notChars Arabic chars those can't be in postfix or prefix
* @param string $preChars Arabic chars those may exists in the prefix
* @param string $postChars Arabic chars those may exists in the postfix
* @param integer $maxPre Max prefix length
* @param integer $maxPost Max postfix length
* @param integer $minStem Min stem length
*
* @return string Arabic stem of the word under giving rules
* @author Khaled Al-Sham'aa <khaled@ar-php.org>
*/
protected static function roughStem (
$word, $notChars, $preChars, $postChars, $maxPre, $maxPost, $minStem
) {
$right = -1;
$left = -1;
$max = mb_strlen($word, 'UTF-8');
for ($i=0; $i < $max; $i++) {
$needle = mb_substr($word, $i, 1, 'UTF-8');
if (mb_strpos($notChars, $needle, 0, 'UTF-8') === false) {
if ($right == -1) {
$right = $i;
}
$left = $i;
}
}
if ($right > $maxPre) {
$right = $maxPre;
}
if ($max - $left - 1 > $maxPost) {
$left = $max - $maxPost -1;
}
for ($i=0; $i < $right; $i++) {
$needle = mb_substr($word, $i, 1, 'UTF-8');
if (mb_strpos($preChars, $needle, 0, 'UTF-8') === false) {
$right = $i;
break;
}
}
for ($i=$max-1; $i>$left; $i--) {
$needle = mb_substr($word, $i, 1, 'UTF-8');
if (mb_strpos($postChars, $needle, 0, 'UTF-8') === false) {
$left = $i;
break;
}
}
if ($left - $right >= $minStem) {
$stem = mb_substr($word, $right, $left-$right+1, 'UTF-8');
} else {
$stem = null;
}
return $stem;
}
}
@@ -0,0 +1,315 @@
<?php
/*
This is a reimplementation in PHP of a simple rule-based stemmer for Croatian
at http://nlp.ffzg.hr/resources/tools/stemmer-for-croatian/ (Python).
The original author is Ivan Pandžić. */
namespace TeamTNT\TNTSearch\Stemmer;
class CroatianStemmer implements Stemmer
{
protected static $stop = ['biti', 'jesam', 'budem', 'sam', 'jesi', 'budeš', 'si', 'jesmo', 'budemo',
'smo', 'jeste', 'budete', 'ste', 'jesu', 'budu', 'su', 'bih', 'bijah', 'bjeh',
'bijaše', 'bi', 'bje', 'bješe', 'bijasmo', 'bismo', 'bjesmo', 'bijaste', 'biste',
'bjeste', 'bijahu', 'biste', 'bjeste', 'bijahu', 'bi', 'biše', 'bjehu', 'bješe',
'bio', 'bili', 'budimo', 'budite', 'bila', 'bilo', 'bile', 'ću', 'ćeš', 'će',
'ćemo', 'ćete', 'želim', 'želiš', 'želi', 'želimo', 'želite', 'žele', 'moram',
'moraš', 'mora', 'moramo', 'morate', 'moraju', 'trebam', 'trebaš', 'treba',
'trebamo', 'trebate', 'trebaju', 'mogu', 'možeš', 'može', 'možemo', 'možete'];
public static function stem($token)
{
if (in_array($token, self::$stop)) {
return $token;
}
return self::korjenuj(self::transformiraj($token));
}
public static function istakniSlogotvornoR($niz)
{
return preg_replace('/(^|[^aeiou])r($|[^aeiou])/', '\1R\2', $niz);
}
public static function imaSamoglasnik($niz)
{
preg_match('/[aeiouR]/', self::istakniSlogotvornoR($niz), $matches);
if (count($matches) > 0) {
return true;
}
return false;
}
public static function transformiraj($pojavnica)
{
foreach (self::$transformations as $trazi => $zamijeni) {
if (self::endsWith($pojavnica, $trazi)) {
return substr($pojavnica, 0, -1 * strlen($trazi)) . $zamijeni;
}
}
return $pojavnica;
}
public static function korjenuj($pojavnica)
{
foreach (self::$rules as $rule) {
$rules = explode(" ", $rule);
$osnova = $rules[0];
$nastavak = $rules[1];
preg_match("/^(" . $osnova . ")(" . $nastavak . ")$/", $pojavnica, $dioba);
if (!empty($dioba)) {
if (self::imaSamoglasnik($dioba[1]) && strlen($dioba[1]) > 1) {
return $dioba[1];
}
}
}
return $pojavnica;
}
public static function endsWith($haystack, $needle)
{
// search forward starting from end minus needle length characters
return $needle === "" || (($temp = strlen($haystack) - strlen($needle)) >= 0 && strpos($haystack, $needle, $temp) !== false);
}
protected static $transformations = [
'lozi' => 'loga',
'lozima' => 'loga',
'pjesi' => 'pjeh',
'pjesima' => 'pjeh',
'vojci' => 'vojka',
'bojci' => 'bojka',
'jaci' => 'jak',
'jacima' => 'jak',
'čajan' => 'čajni',
'ijeran' => 'ijerni',
'laran' => 'larni',
'ijesan' => 'ijesni',
'anjac' => 'anjca',
'ajac' => 'ajca',
'ajaca' => 'ajca',
'ljaca' => 'ljca',
'ljac' => 'ljca',
'ejac' => 'ejca',
'ejaca' => 'ejca',
'ojac' => 'ojca',
'ojaca' => 'ojca',
'ajaka' => 'ajka',
'ojaka' => 'ojka',
'šaca' => 'šca',
'šac' => 'šca',
'inzima' => 'ing',
'inzi' => 'ing',
'tvenici' => 'tvenik',
'tetici' => 'tetika',
'teticima' => 'tetika',
'nstava' => 'nstva',
'nicima' => 'nik',
'ticima' => 'tik',
'zicima' => 'zik',
'snici' => 'snik',
'kuse' => 'kusi',
'kusan' => 'kusni',
'kustava' => 'kustva',
'dušan' => 'dušni',
'antan' => 'antni',
'bilan' => 'bilni',
'tilan' => 'tilni',
'avilan' => 'avilni',
'silan' => 'silni',
'gilan' => 'gilni',
'rilan' => 'rilni',
'nilan' => 'nilni',
'alan' => 'alni',
'ozan' => 'ozni',
'rave' => 'ravi',
'stavan' => 'stavni',
'pravan' => 'pravni',
'tivan' => 'tivni',
'sivan' => 'sivni',
'atan' => 'atni',
'cenata' => 'centa',
'denata' => 'denta',
'genata' => 'genta',
'lenata' => 'lenta',
'menata' => 'menta',
'jenata' => 'jenta',
'venata' => 'venta',
'tetan' => 'tetni',
'pletan' => 'pletni',
'šave' => 'šavi',
'manata' => 'manta',
'tanata' => 'tanta',
'lanata' => 'lanta',
'sanata' => 'santa',
'ačak' => 'ačka',
'ačaka' => 'ačka',
'ušak' => 'uška',
'atak' => 'atka',
'ataka' => 'atka',
'atci' => 'atka',
'atcima' => 'atka',
'etak' => 'etka',
'etaka' => 'etka',
'itak' => 'itka',
'itaka' => 'itka',
'itci' => 'itka',
'otak' => 'otka',
'otaka' => 'otka',
'utak' => 'utka',
'utaka' => 'utka',
'utci' => 'utka',
'utcima' => 'utka',
'eskan' => 'eskna',
'tičan' => 'tični',
'ojsci' => 'ojska',
'esama' => 'esma',
'metara' => 'metra',
'centar' => 'centra',
'centara' => 'centra',
'istara' => 'istra',
'istar' => 'istra',
'ošću' => 'osti',
'daba' => 'dba',
'čcima' => 'čka',
'čci' => 'čka',
'mac' => 'mca',
'maca' => 'mca',
'naca' => 'nca',
'nac' => 'nca',
'voljan' => 'voljni',
'anaka' => 'anki',
'vac' => 'vca',
'vaca' => 'vca',
'saca' => 'sca',
'sac' => 'sca',
'naca' => 'nca',
'nac' => 'nca',
'raca' => 'rca',
'rac' => 'rca',
'aoca' => 'alca',
'alaca' => 'alca',
'alac' => 'alca',
'elaca' => 'elca',
'elac' => 'elca',
'olaca' => 'olca',
'olac' => 'olca',
'olce' => 'olca',
'njac' => 'njca',
'njaca' => 'njca',
'ekata' => 'ekta',
'ekat' => 'ekta',
'izam' => 'izma',
'izama' => 'izma',
'jebe' => 'jebi',
'baci' => 'baci',
'ašan' => 'ašni',
];
protected static $rules = [
".+(s|š)k ijima|ijega|ijemu|ijem|ijim|ijih|ijoj|ijeg|iji|ije|ija|oga|ome|omu|ima|og|om|im|ih|oj|i|e|o|a|u",
".+(s|š)tv ima|om|o|a|u",
// N
".+(t|m|p|r|g)anij ama|ima|om|a|u|e|i| ",
".+an inom|ina|inu|ine|ima|in|om|u|i|a|e| ",
".+in ima|ama|om|a|e|i|u|o| ",
".+on ovima|ova|ove|ovi|ima|om|a|e|i|u| ",
".+n ijima|ijega|ijemu|ijeg|ijem|ijim|ijih|ijoj|iji|ije|ija|iju|ima|ome|omu|oga|oj|om|ih|im|og|o|e|a|u|i| ",
// Ć
".+(a|e|u)ć oga|ome|omu|ega|emu|ima|oj|ih|om|eg|em|og|uh|im|e|a",
// G
".+ugov ima|i|e|a",
".+ug ama|om|a|e|i|u|o",
".+log ama|om|a|u|e| ",
".+[^eo]g ovima|ama|ovi|ove|ova|om|a|e|i|u|o| ",
// I
".+(rrar|ott|ss|ll)i jem|ja|ju|o| ",
// J
".+uj ući|emo|ete|mo|em|eš|e|u| ",
".+(c|č|ć|đ|l|r)aj evima|evi|eva|eve|ama|ima|em|a|e|i|u| ",
".+(b|c|d|l|n|m|ž|g|f|p|r|s|t|z)ij ima|ama|om|a|e|i|u|o| ",
// L
//.+al inom|ina|inu|ine|ima|om|in|i|a|e
//.+[^(lo|ž)]il ima|om|a|e|u|i|
".+[^z]nal ima|ama|om|a|e|i|u|o| ",
".+ijal ima|ama|om|a|e|i|u|o| ",
".+ozil ima|om|a|e|u|i| ",
".+olov ima|i|a|e",
".+ol ima|om|a|u|e|i| ",
// M
".+lem ama|ima|om|a|e|i|u|o| ",
".+ram ama|om|a|e|i|u|o",
//.+(es|e|u)m ama|om|a|e|i|u|o
// R
//.+(a|d|e|o|u)r ama|ima|om|u|a|e|i|
".+(a|d|e|o)r ama|ima|om|u|a|e|i| ",
// S
".+(e|i)s ima|om|e|a|u",
// Š
".+(t|n|j|k|j|t|b|g|v)aš ama|ima|om|em|a|u|i|e| ",
".+(e|i)š ima|ama|om|em|i|e|a|u| ",
// T
".+ikat ima|om|a|e|i|u|o| ",
".+lat ima|om|a|e|i|u|o| ",
".+et ama|ima|om|a|e|i|u|o| ",
//.+ot ama|ima|om|a|u|e|i|
".+(e|i|k|o)st ima|ama|om|a|e|i|u|o| ",
".+išt ima|em|a|e|u",
//.+ut ovima|evima|ove|ovi|ova|eve|evi|eva|ima|om|a|u|e|i|
// V
".+ova smo|ste|hu|ti|še|li|la|le|lo|t|h|o",
".+(a|e|i)v ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|ama|iji|ije|ija|iju|im|ih|oj|om|og|i|a|u|e|o| ",
".+[^dkml]ov ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|iji|ije|ija|iju|im|ih|oj|om|og|i|a|u|e|o| ",
".+(m|l)ov ima|om|a|u|e|i| ",
// PRIDJEVI
".+el ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|iji|ije|ija|iju|im|ih|oj|om|og|i|a|u|e|o| ",
".+(a|e|š)nj ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|iji|ije|ija|iju|ega|emu|eg|em|im|ih|oj|om|og|a|e|i|o|u",
".+čin ama|ome|omu|oga|ima|og|om|im|ih|oj|a|u|i|o|e| ",
".+roši vši|smo|ste|še|mo|te|ti|li|la|lo|le|m|š|t|h|o",
".+oš ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|iji|ije|ija|iju|im|ih|oj|om|og|i|a|u|e| ",
".+(e|o)vit ijima|ijega|ijemu|ijem|ijim|ijih|ijoj|ijeg|iji|ije|ija|oga|ome|omu|ima|og|om|im|ih|oj|i|e|o|a|u| ",
//.+tit ijima|ijega|ijemu|ijem|ijim|ijih|ijoj|ijeg|iji|ije|ija|oga|ome|omu|ima|og|om|im|ih|oj|e|o|a|u|i|
".+ast ijima|ijega|ijemu|ijem|ijim|ijih|ijoj|ijeg|iji|ije|ija|oga|ome|omu|ima|og|om|im|ih|oj|i|e|o|a|u| ",
".+k ijemu|ijima|ijega|ijeg|ijem|ijim|ijih|ijoj|oga|ome|omu|ima|iji|ije|ija|iju|im|ih|oj|om|og|i|a|u|e|o| ",
// GLAGOLI
".+(e|a|i|u)va jući|smo|ste|jmo|jte|ju|la|le|li|lo|mo|na|ne|ni|no|te|ti|še|hu|h|j|m|n|o|t|v|š| ",
".+ir ujemo|ujete|ujući|ajući|ivat|ujem|uješ|ujmo|ujte|avši|asmo|aste|ati|amo|ate|aju|aše|ahu|ala|alo|ali|ale|uje|uju|uj|al|an|am|aš|at|ah|ao",
".+ač ismo|iste|iti|imo|ite|iše|eći|ila|ilo|ili|ile|ena|eno|eni|ene|io|im|iš|it|ih|en|i|e",
".+ača vši|smo|ste|smo|ste|hu|ti|mo|te|še|la|lo|li|le|ju|na|no|ni|ne|o|m|š|t|h|n",
//.+ači smo|ste|ti|li|la|lo|le|mo|te|še|m|š|t|h|o|
// Druga_vrsta
".+n uvši|usmo|uste|ući|imo|ite|emo|ete|ula|ulo|ule|uli|uto|uti|uta|em|eš|uo|ut|e|u|i",
".+ni vši|smo|ste|ti|mo|te|mo|te|la|lo|le|li|m|š|o",
// A
".+((a|r|i|p|e|u)st|[^o]g|ik|uc|oj|aj|lj|ak|ck|čk|šk|uk|nj|im|ar|at|et|št|it|ot|ut|zn|zv)a jući|vši|smo|ste|jmo|jte|jem|mo|te|je|ju|ti|še|hu|la|li|le|lo|na|no|ni|ne|t|h|o|j|n|m|š",
".+ur ajući|asmo|aste|ajmo|ajte|amo|ate|aju|ati|aše|ahu|ala|ali|ale|alo|ana|ano|ani|ane|al|at|ah|ao|aj|an|am|aš",
".+(a|i|o)staj asmo|aste|ahu|ati|emo|ete|aše|ali|ući|ala|alo|ale|mo|ao|em|eš|at|ah|te|e|u| ",
".+(b|c|č|ć|d|e|f|g|j|k|n|r|t|u|v)a lama|lima|lom|lu|li|la|le|lo|l",
".+(t|č|j|ž|š)aj evima|evi|eva|eve|ama|ima|em|a|e|i|u| ",
//.+(e|j|k|r|u|v)al ama|ima|om|u|i|a|e|o|
//.+(e|j|k|r|t|u|v)al ih|im
".+([^o]m|ič|nč|uč|b|c|ć|d|đ|h|j|k|l|n|p|r|s|š|v|z|ž)a jući|vši|smo|ste|jmo|jte|mo|te|ju|ti|še|hu|la|li|le|lo|na|no|ni|ne|t|h|o|j|n|m|š",
".+(a|i|o)sta dosmo|doste|doše|nemo|demo|nete|dete|nimo|nite|nila|vši|nem|dem|neš|deš|doh|de|ti|ne|nu|du|la|li|lo|le|t|o",
".+ta smo|ste|jmo|jte|vši|ti|mo|te|ju|še|la|lo|le|li|na|no|ni|ne|n|j|o|m|š|t|h",
".+inj asmo|aste|ati|emo|ete|ali|ala|alo|ale|aše|ahu|em|eš|at|ah|ao",
".+as temo|tete|timo|tite|tući|tem|teš|tao|te|li|ti|la|lo|le",
// I
".+(elj|ulj|tit|ac|ič|od|oj|et|av|ov)i vši|eći|smo|ste|še|mo|te|ti|li|la|lo|le|m|š|t|h|o",
".+(tit|jeb|ar|ed|uš|ič)i jemo|jete|jem|ješ|smo|ste|jmo|jte|vši|mo|še|te|ti|ju|je|la|lo|li|le|t|m|š|h|j|o",
".+(b|č|d|l|m|p|r|s|š|ž)i jemo|jete|jem|ješ|smo|ste|jmo|jte|vši|mo|lu|še|te|ti|ju|je|la|lo|li|le|t|m|š|h|j|o",
".+luč ujete|ujući|ujemo|ujem|uješ|ismo|iste|ujmo|ujte|uje|uju|iše|iti|imo|ite|ila|ilo|ili|ile|ena|eno|eni|ene|uj|io|en|im|iš|it|ih|e|i",
".+jeti smo|ste|še|mo|te|ti|li|la|lo|le|m|š|t|h|o",
".+e lama|lima|lom|lu|li|la|le|lo|l",
".+i lama|lima|lom|lu|li|la|le|lo|l",
// Pridjev_t
".+at ijega|ijemu|ijima|ijeg|ijem|ijih|ijim|ima|oga|ome|omu|iji|ije|ija|iju|oj|og|om|im|ih|a|u|i|e|o| ",
// Pridjev
".+et avši|ući|emo|imo|em|eš|e|u|i",
".+ ajući|alima|alom|avši|asmo|aste|ajmo|ajte|ivši|amo|ate|aju|ati|aše|ahu|ali|ala|ale|alo|ana|ano|ani|ane|am|aš|at|ah|ao|aj|an",
".+ anje|enje|anja|enja|enom|enoj|enog|enim|enih|anom|anoj|anog|anim|anih|eno|ovi|ova|oga|ima|ove|enu|anu|ena|ama",
".+ nijega|nijemu|nijima|nijeg|nijem|nijim|nijih|nima|niji|nije|nija|niju|noj|nom|nog|nim|nih|an|na|nu|ni|ne|no",
".+ om|og|im|ih|em|oj|an|u|o|i|e|a",
];
}
@@ -0,0 +1,693 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
*
* @link http://snowball.tartarus.org/algorithms/french/stemmer.html
* The original author is wamania
*
*/
class FrenchStemmer implements Stemmer
{
/**
* All french vowels
*/
protected static $vowels = ['a', 'e', 'i', 'o', 'u', 'y', 'â', 'à', 'ë', 'é', 'ê', 'è', 'ï', 'î', 'ô', 'û', 'ù'];
protected $word;
/**
* helper, contains stringified list of vowels
* @var string
*/
protected $plainVowels;
/**
* The original word, use to check if word has been modified
* @var string
*/
protected $originalWord;
/**
* RV value
* @var string
*/
protected $rv;
/**
* RV index (based on the beginning of the word)
* @var int
*/
protected $rvIndex;
/**
* R1 value
* @var int
*/
protected $r1;
/**
* R1 index (based on the beginning of the word)
* @var int
*/
protected $r1Index;
/**
* R2 value
* @var int
*/
protected $r2;
/**
* R2 index (based on the beginning of the word)
* @var int
*/
protected $r2Index;
public static function stem($word)
{
return (new static)->analyze($word);
}
public function analyze($word)
{
$this->word = mb_strtolower($word);
$this->plainVowels = implode('', static::$vowels);
$this->step0();
$this->rv();
$this->r1();
$this->r2();
// to know if step1, 2a or 2b have altered the word
$this->originalWord = $this->word;
$nextStep = $this->step1();
// Do step 2a if either no ending was removed by step 1, or if one of endings amment, emment, ment, ments was found.
if (($nextStep == 2) || ($this->originalWord === $this->word) ) {
$modified = $this->step2a();
if (!$modified) {
$this->step2b();
}
}
if ($this->word != $this->originalWord) {
$this->step3();
} else {
$this->step4();
}
$this->step5();
$this->step6();
$this->finish();
return $this->word;
}
/**
* Assume the word is in lower case.
* Then put into upper case u or i preceded and followed by a vowel, and y preceded or followed by a vowel.
* u after q is also put into upper case. For example,
* jouer -> joUer
* ennuie -> ennuIe
* yeux -> Yeux
* quand -> qUand
*/
private function step0()
{
$this->word = preg_replace('#([q])u#u', '$1U', $this->word);
$this->word = preg_replace('#(['.$this->plainVowels.'])y#u', '$1Y', $this->word);
$this->word = preg_replace('#y(['.$this->plainVowels.'])#u', 'Y$1', $this->word);
$this->word = preg_replace('#(['.$this->plainVowels.'])u(['.$this->plainVowels.'])#u', '$1U$2', $this->word);
$this->word = preg_replace('#(['.$this->plainVowels.'])i(['.$this->plainVowels.'])#u', '$1I$2', $this->word);
}
/**
* Step 1
* Search for the longest among the following suffixes, and perform the action indicated.
*
* @return integer Next step number
*/
private function step1()
{
// ance iqUe isme able iste eux ances iqUes ismes ables istes
// delete if in R2
if (($position = $this->search([
'ances', 'iqUes', 'ismes', 'ables', 'istes', 'ance', 'iqUe','isme', 'able', 'iste', 'eux'
])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
}
return 3;
}
// atrice ateur ation atrices ateurs ations
// delete if in R2
// if preceded by ic, delete if in R2, else replace by iqU
if (($position = $this->search(['atrices', 'ateurs', 'ations', 'atrice', 'ateur', 'ation'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
if (($position2 = $this->searchIfInR2(['ic'])) !== false) {
$this->word = mb_substr($this->word, 0, $position2);
} else {
$this->word = preg_replace('#(ic)$#u', 'iqU', $this->word);
}
}
return 3;
}
// logie logies
// replace with log if in R2
if (($position = $this->search(['logies', 'logie'])) !== false) {
if ($this->inR2($position)) {
$this->word = preg_replace('#(logies|logie)$#u', 'log', $this->word);
}
return 3;
}
// usion ution usions utions
// replace with u if in R2
if (($position = $this->search(['usions', 'utions', 'usion', 'ution'])) !== false) {
if ($this->inR2($position)) {
$this->word = preg_replace('#(usion|ution|usions|utions)$#u', 'u', $this->word);
}
return 3;
}
// ence ences
// replace with ent if in R2
if (($position = $this->search(['ences', 'ence'])) !== false) {
if ($this->inR2($position)) {
$this->word = preg_replace('#(ence|ences)$#u', 'ent', $this->word);
}
return 3;
}
// issement issements
// delete if in R1 and preceded by a non-vowel
if (($position = $this->search(['issements', 'issement'])) != false) {
if ($this->inR1($position)) {
$before = $position - 1;
$letter = mb_substr($this->word, $before, 1);
if (! in_array($letter, static::$vowels)) {
$this->word = mb_substr($this->word, 0, $position);
}
}
return 3;
}
// ement ements
// delete if in RV
// if preceded by iv, delete if in R2 (and if further preceded by at, delete if in R2), otherwise,
// if preceded by eus, delete if in R2, else replace by eux if in R1, otherwise,
// if preceded by abl or iqU, delete if in R2, otherwise,
// if preceded by ièr or Ièr, replace by i if in RV
if (($position = $this->search(['ements', 'ement'])) !== false) {
if ($this->inRv($position)) {
$this->word = mb_substr($this->word, 0, $position);
}
if (($position = $this->searchIfInR2(['iv'])) !== false) {
$this->word = mb_substr($this->word, 0, $position);
if (($position2 = $this->searchIfInR2(['at'])) !== false) {
$this->word = mb_substr($this->word, 0, $position2);
}
} elseif (($position = $this->search(['eus'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
} elseif ($this->inR1($position)) {
$this->word = preg_replace('#(eus)$#u', 'eux', $this->word);
}
} elseif (($position = $this->searchIfInR2(['abl', 'iqU'])) !== false) {
$this->word = mb_substr($this->word, 0, $position);
} elseif (($this->searchIfInRv(['ièr', 'Ièr'])) !== false) {
$this->word = preg_replace('#(ièr|Ièr)$#u', 'i', $this->word);
}
return 3;
}
// ité ités
// delete if in R2
// if preceded by abil, delete if in R2, else replace by abl, otherwise,
// if preceded by ic, delete if in R2, else replace by iqU, otherwise,
// if preceded by iv, delete if in R2
if (($position = $this->search(['ités', 'ité'])) !== false) {
// delete if in R2
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
}
// if preceded by abil, delete if in R2, else replace by abl, otherwise,
if (($position = $this->search(['abil'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
} else {
$this->word = preg_replace('#(abil)$#u', 'abl', $this->word);
}
// if preceded by ic, delete if in R2, else replace by iqU, otherwise,
} elseif (($position = $this->search(['ic'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
} else {
$this->word = preg_replace('#(ic)$#u', 'iqU', $this->word);
}
// if preceded by iv, delete if in R2
} elseif (($position = $this->searchIfInR2(['iv'])) !== false) {
$this->word = mb_substr($this->word, 0, $position);
}
return 3;
}
// if ive ifs ives
// delete if in R2
// if preceded by at, delete if in R2 (and if further preceded by ic, delete if in R2, else replace by iqU)
if (($position = $this->search(['ifs', 'ives', 'if', 'ive'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
}
if (($position = $this->searchIfInR2(['at'])) !== false) {
$this->word = mb_substr($this->word, 0, $position);
if (($position2 = $this->search(['ic'])) !== false) {
if ($this->inR2($position2)) {
$this->word = mb_substr($this->word, 0, $position2);
} else {
$this->word = preg_replace('#(ic)$#u', 'iqU', $this->word);
}
}
}
return 3;
}
// eaux
// replace with eau
if (($this->search(['eaux'])) !== false) {
$this->word = preg_replace('#(eaux)$#u', 'eau', $this->word);
return 3;
}
// aux
// replace with al if in R1
if (($position = $this->search(['aux'])) !== false) {
if ($this->inR1($position)) {
$this->word = preg_replace('#(aux)$#u', 'al', $this->word);
}
return 3;
}
// euse euses
// delete if in R2, else replace by eux if in R1
if (($position = $this->search(['euses', 'euse'])) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
} elseif ($this->inR1($position)) {
$this->word = preg_replace('#(euses|euse)$#u', 'eux', $this->word);
}
return 3;
}
// amment
// replace with ant if in RV
if ( ($position = $this->search(['amment'])) !== false) {
if ($this->inRv($position)) {
$this->word = preg_replace('#(amment)$#u', 'ant', $this->word);
}
return 2;
}
// emment
// replace with ent if in RV
if (($position = $this->search(['emment'])) !== false) {
if ($this->inRv($position)) {
$this->word = preg_replace('#(emment)$#u', 'ent', $this->word);
}
return 2;
}
// ment ments
// delete if preceded by a vowel in RV
if (($position = $this->search(['ments', 'ment'])) != false) {
$before = $position - 1;
$letter = mb_substr($this->word, $before, 1);
if ($this->inRv($before) && (in_array($letter, static::$vowels)) ) {
$this->word = mb_substr($this->word, 0, $position);
}
return 2;
}
return 2;
}
/**
* Step 2a: Verb suffixes beginning i
* In steps 2a and 2b all tests are confined to the RV region.
* Search for the longest among the following suffixes and if found, delete if preceded by a non-vowel.
* îmes ît îtes i ie ies ir ira irai iraIent irais irait iras irent irez iriez
* irions irons iront is issaIent issais issait issant issante issantes issants isse
* issent isses issez issiez issions issons it
* (Note that the non-vowel itself must also be in RV.)
*/
private function step2a()
{
if (($position = $this->searchIfInRv([
'îmes', 'îtes', 'ît', 'ies', 'ie', 'iraIent', 'irais', 'irait', 'irai', 'iras', 'ira', 'irent', 'irez', 'iriez',
'irions', 'irons', 'iront', 'ir', 'issaIent', 'issais', 'issait', 'issant', 'issantes', 'issante', 'issants',
'issent', 'isses', 'issez', 'isse', 'issiez', 'issions', 'issons', 'is', 'it', 'i'])) !== false) {
$before = $position - 1;
$letter = mb_substr($this->word, $before, 1);
if ( $this->inRv($before) && (!in_array($letter, static::$vowels)) ) {
$this->word = mb_substr($this->word, 0, $position);
return true;
}
}
return false;
}
/**
* Do step 2b if step 2a was done, but failed to remove a suffix.
* Step 2b: Other verb suffixes
*/
private function step2b()
{
// é ée ées és èrent er era erai eraIent erais erait eras erez eriez erions erons eront ez iez
// delete
if (($position = $this->searchIfInRv([
'ées', 'èrent', 'erais', 'erait', 'erai', 'eraIent', 'eras', 'erez', 'eriez',
'erions', 'erons', 'eront', 'era', 'er', 'iez', 'ez','és', 'ée', 'é'])) !== false) {
$this->word = mb_substr($this->word, 0, $position);
return true;
}
// âmes ât âtes a ai aIent ais ait ant ante antes ants as asse assent asses assiez assions
// delete
// if preceded by e, delete
if (($position = $this->searchIfInRv([
'âmes', 'âtes', 'ât', 'aIent', 'ais', 'ait', 'antes', 'ante', 'ants', 'ant',
'assent', 'asses', 'assiez', 'assions', 'asse', 'as', 'ai', 'a'])) !== false) {
$before = $position - 1;
$letter = mb_substr($this->word, $before, 1);
if ( $this->inRv($before) && ($letter === 'e') ) {
$this->word = mb_substr($this->word, 0, $before);
} else {
$this->word = mb_substr($this->word, 0, $position);
}
return true;
}
// ions
// delete if in R2
if ( ($position = $this->searchIfInRv(array('ions'))) !== false) {
if ($this->inR2($position)) {
$this->word = mb_substr($this->word, 0, $position);
}
return true;
}
return false;
}
/**
* Step 3: Replace final Y with i or final ç with c
*/
private function step3()
{
$this->word = preg_replace('#(Y)$#u', 'i', $this->word);
$this->word = preg_replace('#(ç)$#u', 'c', $this->word);
}
/**
* Step 4: Residual suffix
*/
private function step4()
{
//If the word ends s, not preceded by a, i, o, u, è or s, delete it.
if (preg_match('#[^aiouès]s$#', $this->word)) {
$this->word = mb_substr($this->word, 0, -1);
}
// In the rest of step 4, all tests are confined to the RV region.
// ion
// delete if in R2 and preceded by s or t
if ((($position = $this->searchIfInRv(['ion'])) !== false) && ($this->inR2($position)) ) {
$before = $position - 1;
$letter = mb_substr($this->word, $before, 1);
if ( $this->inRv($before) && (($letter === 's') || ($letter === 't')) ) {
$this->word = mb_substr($this->word, 0, $position);
}
return true;
}
// ier ière Ier Ière
// replace with i
if (($this->searchIfInRv(['ier', 'ière', 'Ier', 'Ière'])) !== false) {
$this->word = preg_replace('#(ier|ière|Ier|Ière)$#u', 'i', $this->word);
return true;
}
// e
// delete
if (($this->searchIfInRv(['e'])) !== false) {
$this->word = mb_substr($this->word, 0, -1);
return true;
}
// ë
// if preceded by gu, delete
if (($position = $this->searchIfInRv(['guë'])) !== false) {
if ($this->inRv($position + 2)) {
$this->word = mb_substr($this->word, 0, -1);
return true;
}
}
return false;
}
/**
* Step 5: Undouble
* If the word ends enn, onn, ett, ell or eill, delete the last letter
*/
private function step5()
{
if ($this->search(['enn', 'onn', 'ett', 'ell', 'eill']) !== false) {
$this->word = mb_substr($this->word, 0, -1);
}
}
/**
* Step 6: Un-accent
* If the words ends é or è followed by at least one non-vowel, remove the accent from the e.
*/
private function step6()
{
$this->word = preg_replace('#(é|è)([^'.$this->plainVowels.']+)$#u', 'e$2', $this->word);
}
/**
* And finally:
* Turn any remaining I, U and Y letters in the word back into lower case.
*/
private function finish()
{
$this->word = str_replace(['I','U','Y'], ['i', 'u', 'y'], $this->word);
}
/**
* If the word begins with two vowels, RV is the region after the third letter,
* otherwise the region after the first vowel not at the beginning of the word,
* or the end of the word if these positions cannot be found.
* (Exceptionally, par, col or tap, at the begining of a word is also taken to define RV as the region to their right.)
*/
protected function rv()
{
$length = mb_strlen($this->word);
$this->rv = '';
$this->rvIndex = $length;
if ($length < 3) {
return true;
}
// If the word begins with two vowels, RV is the region after the third letter
$first = mb_substr($this->word, 0, 1);
$second = mb_substr($this->word, 1, 1);
if ( (in_array($first, static::$vowels)) && (in_array($second, static::$vowels)) ) {
$this->rv = mb_substr($this->word, 3);
$this->rvIndex = 3;
return true;
}
// (Exceptionally, par, col or tap, at the begining of a word is also taken to define RV as the region to their right.)
$begin3 = mb_substr($this->word, 0, 3);
if (in_array($begin3, ['par', 'col', 'tap'])) {
$this->rv = mb_substr($this->word, 3);
$this->rvIndex = 3;
return true;
}
// otherwise the region after the first vowel not at the beginning of the word,
for ($i = 1; $i < $length; ++$i) {
$letter = mb_substr($this->word, $i, 1);
if (in_array($letter, static::$vowels)) {
$this->rv = mb_substr($this->word, ($i + 1));
$this->rvIndex = $i + 1;
return true;
}
}
return false;
}
protected function inRv($position)
{
return ($position >= $this->rvIndex);
}
protected function inR1($position)
{
return ($position >= $this->r1Index);
}
protected function inR2($position)
{
return ($position >= $this->r2Index);
}
protected function searchIfInRv($suffixes)
{
return $this->search($suffixes, $this->rvIndex);
}
protected function searchIfInR2($suffixes)
{
return $this->search($suffixes, $this->r2Index);
}
protected function search($suffixes, $offset = 0)
{
$length = mb_strlen($this->word);
if ($offset > $length) {
return false;
}
foreach ($suffixes as $suffixe) {
if ((($position = mb_strrpos($this->word, $suffixe, $offset)) !== false)
&& ((mb_strlen($suffixe) + $position) == $length)) {
return $position;
}
}
return false;
}
/**
* R1 is the region after the first non-vowel following a vowel, or the end of the word if there is no such non-vowel.
*/
protected function r1()
{
list($this->r1Index, $this->r1) = $this->rx($this->word);
}
/**
* R2 is the region after the first non-vowel following a vowel in R1, or the end of the word if there is no such non-vowel.
*/
protected function r2()
{
list($index, $value) = $this->rx($this->r1);
$this->r2 = $value;
$this->r2Index = $this->r1Index + $index;
}
/**
* Common function for R1 and R2
* Search the region after the first non-vowel following a vowel in $word, or the end of the word if there is no such non-vowel.
* R1 : $in = $this->word
* R2 : $in = R1
*/
protected function rx($in)
{
$length = mb_strlen($in);
// defaults
$value = '';
$index = $length;
// we search all vowels
$vowels = [];
for ($i = 0; $i < $length; ++$i) {
$letter = mb_substr($in, $i, 1);
if (in_array($letter, static::$vowels)) {
$vowels[] = $i;
}
}
// search the non-vowel following a vowel
foreach ($vowels as $position) {
$after = $position + 1;
$letter = mb_substr($in, $after, 1);
if (!in_array($letter, static::$vowels)) {
$index = $after + 1;
$value = mb_substr($in, ($after + 1));
break;
}
}
return [$index, $value];
}
}
@@ -0,0 +1,248 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
* Copyright (c) 2013 Aris Buzachis (buzachis.aris@gmail.com)
*
* All rights reserved.
*
* This script is free software.
*
* DISCLAIMER:
*
* IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
* (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON
* ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
* SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*/
/**
* Takes a word and reduces it to its German stem using the Porter stemmer algorithm.
*
* References:
* - http://snowball.tartarus.org/algorithms/porter/stemmer.html
* - http://snowball.tartarus.org/algorithms/german/stemmer.html
*
* Usage:
* $stem = GermanStemmer::stem($word);
*
* @author Aris Buzachis <buzachis.aris@gmail.com>
* @author Pascal Landau <kontakt@myseosolution.de>
*/
class GermanStemmer implements Stemmer
{
/**
* R1 and R2 regions (see the Porter algorithm)
*/
private static $R1;
private static $R2;
private static $cache = array();
private static $vowels = array('a', 'e', 'i', 'o', 'u', 'y', 'ä', 'ö', 'ü');
private static $s_ending = array('b', 'd', 'f', 'g', 'h', 'k', 'l', 'm', 'n', 'r', 't');
private static $st_ending = array('b', 'd', 'f', 'g', 'h', 'k', 'l', 'm', 'n', 't');
/**
* Gets the stem of $word.
* @param string $word
* @return string
*/
public static function stem($word)
{
$word = mb_strtolower($word);
//check for invalid characters
preg_match("#.#u", $word);
if (preg_last_error() !== 0) {
throw new \InvalidArgumentException("Word '$word' seems to be errornous. Error code from preg_last_error(): " . preg_last_error());
}
if (!isset(self::$cache[$word])) {
$result = self::getStem($word);
self::$cache[$word] = $result;
}
return self::$cache[$word];
}
/**
* @param $word
* @return string
*/
private static function getStem($word)
{
$word = self::step0a($word);
$word = self::step1($word);
$word = self::step2($word);
$word = self::step3($word);
$word = self::step0b($word);
return $word;
}
/**
* Replaces to protect some characters
* @param string $word
* @return string mixed
*/
private static function step0a($word)
{
$vstr = implode('', self::$vowels);
$word = preg_replace('#([' . $vstr . '])u([' . $vstr . '])#u', '$1U$2', $word);
$word = preg_replace('#([' . $vstr . '])y([' . $vstr . '])#u', '$1Y$2', $word);
return $word;
}
/**
* Undo the initial replaces
* @param string $word
* @return string
*/
private static function step0b($word)
{
$word = str_replace(array('ä', 'ö', 'ü', 'U', 'Y'), array('a', 'o', 'u', 'u', 'y'), $word);
return $word;
}
private static function step1($word)
{
$word = str_replace('ß', 'ss', $word);
self::getR($word);
$replaceCount = 0;
$arr = array('em', 'ern', 'er');
foreach ($arr as $s) {
self::$R1 = preg_replace('#' . $s . '$#u', '', self::$R1, -1, $replaceCount);
if ($replaceCount > 0) {
$word = preg_replace('#' . $s . '$#u', '', $word);
}
}
$arr = array('en', 'es', 'e');
foreach ($arr as $s) {
self::$R1 = preg_replace('#' . $s . '$#u', '', self::$R1, -1, $replaceCount);
if ($replaceCount > 0) {
$word = preg_replace('#' . $s . '$#u', '', $word);
$word = preg_replace('#niss$#u', 'nis', $word);
}
}
$word = preg_replace('/([' . implode('', self::$s_ending) . '])s$/u', '$1', $word);
return $word;
}
private static function step2($word)
{
self::getR($word);
$replaceCount = 0;
$arr = array('est', 'er', 'en');
foreach ($arr as $s) {
self::$R1 = preg_replace('#' . $s . '$#u', '', self::$R1, -1, $replaceCount);
if ($replaceCount > 0) {
$word = preg_replace('#' . $s . '$#u', '', $word);
}
}
if (strpos(self::$R1, 'st') !== false) {
self::$R1 = preg_replace('#st$#u', '', self::$R1);
$word = preg_replace('#(...[' . implode('', self::$st_ending) . '])st$#u', '$1', $word);
}
return $word;
}
private static function step3($word)
{
self::getR($word);
$replaceCount = 0;
$arr = array('end', 'ung');
foreach ($arr as $s) {
if (preg_match('#' . $s . '$#u', self::$R2)) {
$word = preg_replace('#([^e])' . $s . '$#u', '$1', $word, -1, $replaceCount);
if ($replaceCount > 0) {
self::$R2 = preg_replace('#' . $s . '$#u', '', self::$R2, -1, $replaceCount);
}
}
}
$arr = array('isch', 'ik', 'ig');
foreach ($arr as $s) {
if (preg_match('#' . $s . '$#u', self::$R2)) {
$word = preg_replace('#([^e])' . $s . '$#u', '$1', $word, -1, $replaceCount);
if ($replaceCount > 0) {
self::$R2 = preg_replace('#' . $s . '$#u', '', self::$R2);
}
}
}
$arr = array('lich', 'heit');
foreach ($arr as $s) {
self::$R2 = preg_replace('#' . $s . '$#u', '', self::$R2, -1, $replaceCount);
if ($replaceCount > 0) {
$word = preg_replace('#' . $s . '$#u', '', $word);
} else {
if (preg_match('#' . $s . '$#u', self::$R1)) {
$word = preg_replace('#(er|en)' . $s . '$#u', '$1', $word, -1, $replaceCount);
if ($replaceCount > 0) {
self::$R1 = preg_replace('#' . $s . '$#u', '', self::$R1);
}
}
}
}
$arr = array('keit');
foreach ($arr as $s) {
self::$R2 = preg_replace('#' . $s . '$#u', '', self::$R2, -1, $replaceCount);
if ($replaceCount > 0) {
$word = preg_replace('#' . $s . '$#u', '', $word);
}
}
return $word;
}
/**
* Find R1 and R2
* @param string $word
*/
private static function getR($word)
{
self::$R1 = "";
self::$R2 = "";
$vowels = implode("", self::$vowels);
$vowelGroup = "[{$vowels}]";
$nonVowelGroup = "[^{$vowels}]";
// R1 is the region after the first non-vowel following a vowel, or is the null region at the end of the word if there is no such non-vowel.
$pattern = "#(?P<rest>.*?{$vowelGroup}{$nonVowelGroup})(?P<r>.*)#u";
if (preg_match($pattern, $word, $match)) {
$rest = $match["rest"];
$r1 = $match["r"];
// [...], but then R1 is adjusted so that the region before it contains at least 3 letters.
$cutOff = 3 - mb_strlen($rest);
if ($cutOff > 0) {
$r1 = mb_substr($r1, $cutOff);
}
self::$R1 = $r1;
}
//R2 is the region after the first non-vowel following a vowel in R1, or is the null region at the end of the word if there is no such non-vowel.
if (preg_match($pattern, self::$R1, $match)) {
self::$R2 = $match["r"];
}
}
}
@@ -0,0 +1,451 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/*
* The following code, downloaded from <https://www.drupal.org/project/italianstemmer>,
* was originally written by Roberto Mirizzi (<roberto.mirizzi@gmail.com>,
* <http://sisinflab.poliba.it/mirizzi/>) in February 2007. It was the PHP5 implementation
* of Martin Porter's stemming algorithm for Italian language. This algorithm can be found
* at the address: <http://snowball.tartarus.org/algorithms/italian/stemmer.html>.
*
* It was rewritten in March 2017 for TNTSearch by GaspariLab S.r.l., <dev@gasparilab.it>.
*/
/*
* This program is free software: you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation, either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program. If not, see <http://www.gnu.org/licenses/>.
*/
class ItalianStemmer implements Stemmer
{
private static $cache = [];
private static $vocali = ['a', 'e', 'i', 'o', 'u', 'à', 'è', 'ì', 'ò', 'ù'];
private static $consonanti = [
'b', 'c', 'd', 'f', 'g', 'h', 'j', 'k', 'l', 'm', 'n', 'p', 'q', 'r', 's', 't', 'v', 'w', 'x', 'y', 'z',
'I', 'U',
];
private static $accenti_acuti = ['á', 'é', 'í', 'ó', 'ú'];
private static $accenti_gravi = ['à', 'è', 'ì', 'ò', 'ù'];
private static $suffissi_step0 = [
'ci', 'gli', 'la', 'le', 'li', 'lo', 'mi', 'ne', 'si', 'ti', 'vi', 'sene',
'gliela', 'gliele', 'glieli', 'glielo', 'gliene', 'mela', 'mele', 'meli', 'melo', 'mene', 'tela', 'tele',
'teli', 'telo', 'tene', 'cela', 'cele', 'celi', 'celo', 'cene', 'vela', 'vele', 'veli', 'velo', 'vene',
];
private static $suffissi_step1_a = [
'anza', 'anze', 'ico', 'ici', 'ica', 'ice', 'iche', 'ichi', 'ismo', 'ismi', 'abile', 'abili', 'ibile',
'ibili', 'ista', 'iste', 'isti', 'istà', 'istè', 'istì', 'oso', 'osi', 'osa', 'ose', 'mente', 'atrice',
'atrici', 'ante', 'anti',
];
private static $suffissi_step1_b = ['azione', 'azioni', 'atore', 'atori'];
private static $suffissi_step1_c = ['logia', 'logie'];
private static $suffissi_step1_d = ['uzione', 'uzioni', 'usione', 'usioni'];
private static $suffissi_step1_e = ['enza', 'enze'];
private static $suffissi_step1_f = ['amento', 'amenti', 'imento', 'imenti'];
private static $suffissi_step1_g = ['amente'];
private static $suffissi_step1_h = ['ità'];
private static $suffissi_step1_i = ['ivo', 'ivi', 'iva', 'ive'];
private static $suffissi_step2 = [
'ammo', 'ando', 'ano', 'are', 'arono', 'asse', 'assero', 'assi', 'assimo', 'ata', 'ate', 'ati', 'ato', 'ava',
'avamo', 'avano', 'avate', 'avi', 'avo', 'emmo', 'enda', 'ende', 'endi', 'endo', 'erà', 'erai', 'eranno',
'ere', 'erebbe', 'erebbero', 'erei', 'eremmo', 'eremo', 'ereste', 'eresti', 'erete', 'erò', 'erono', 'essero',
'ete', 'eva', 'evamo', 'evano', 'evate', 'evi', 'evo', 'Yamo', 'iamo', 'immo', 'irà', 'irai', 'iranno', 'ire',
'irebbe', 'irebbero', 'irei', 'iremmo', 'iremo', 'ireste', 'iresti', 'irete', 'irò', 'irono', 'isca',
'iscano', 'isce', 'isci', 'isco', 'iscono', 'issero', 'ita', 'ite', 'iti', 'ito', 'iva', 'ivamo', 'ivano',
'ivate', 'ivi', 'ivo', 'ono', 'uta', 'ute', 'uti', 'uto', 'ar', 'ir',
];
private static $ante_suff_a = ['ando', 'endo'];
private static $ante_suff_b = ['ar', 'er', 'ir'];
public function __construct()
{
usort(self::$suffissi_step0, function($a,$b) { return mb_strlen($a)>mb_strlen($b) ? -1 : 1; });
usort(self::$suffissi_step1_a, function($a,$b) { return mb_strlen($a)>mb_strlen($b) ? -1 : 1;});
usort(self::$suffissi_step2, function($a,$b) { return mb_strlen($a)>mb_strlen($b) ? -1 : 1;});
}
/**
* Gets the stem of $word.
*
* @param string $word
*
* @return string
*/
public static function stem($word)
{
$word = mb_strtolower($word);
// Check for invalid characters
preg_match('#.#u', $word);
if (preg_last_error() !== 0) {
throw new \InvalidArgumentException('Word "'.$word.'" seems to be errornous.
Error code from preg_last_error(): '.preg_last_error());
}
if (!isset(self::$cache[$word])) {
$result = self::getStem($word);
self::$cache[$word] = $result;
}
return self::$cache[$word];
}
/**
* @param $word
*
* @return string
*/
private static function getStem($word)
{
$str = self::trim($word);
$str = self::toLower($str);
$str = self::replaceAccAcuti($str);
$str = self::putUAfterQToUpper($str);
$str = self::IUBetweenVowToUpper($str);
$step0 = self::step0($str);
$step1 = self::step1($step0);
$step2 = self::step2($step0, $step1);
$step3a = self::step3a($step2);
$step3b = self::step3b($step3a);
$step4 = self::step4($step3b);
return $step4;
}
private static function trim($str)
{
return trim($str);
}
private static function toLower($str)
{
return strtolower($str);
}
private static function replaceAccAcuti($str)
{
return str_replace(self::$accenti_acuti, self::$accenti_gravi, $str); //strtr
}
private static function putUAfterQToUpper($str)
{
return str_replace('qu', 'qU', $str);
}
private static function IUBetweenVowToUpper($str)
{
$pattern = '/([aeiouàèìòù])([iu])([aeiouàèìòù])/';
return preg_replace_callback($pattern, function ($matches) {
return strtoupper($matches[0]);
}, $str);
}
private static function returnRV($str)
{
/*
If the second letter is a consonant, RV is the region after the next following vowel,
or if the first two letters are vowels, RV is the region after the next consonant, and otherwise
(consonant-vowel case) RV is the region after the third letter.
But RV is the end of the word if these positions cannot be found. Example:
m a c h o [ho] o l i v a [va] t r a b a j o [bajo] á u r e o [eo] prezzo sprezzante
*/
if (mb_strlen($str) < 2) {
return '';
} //$str;
if (in_array($str[1], self::$consonanti)) {
$str = mb_substr($str, 2);
$str = strpbrk($str, implode(self::$vocali));
return mb_substr($str, 1); //secondo me devo mettere 1
} elseif (in_array($str[0], self::$vocali) && in_array($str[1], self::$vocali)) {
$str = strpbrk($str, implode(self::$consonanti));
return mb_substr($str, 1);
} elseif (in_array($str[0], self::$consonanti) && in_array($str[1], self::$vocali)) {
return mb_substr($str, 3);
}
}
private static function returnR1($str)
{
/*
R1 is the region after the first non-vowel following a vowel, or is the null region at the end
of the word if there is no such non-vowel. Example:
beautiful [iful] beauty [y] beau [NULL] animadversion [imadversion] sprinkled [kled] eucharist [harist]
*/
$pattern = '/['.implode(self::$vocali).']+'.'['.implode(self::$consonanti).']'.'(.*)/';
preg_match($pattern, $str, $matches);
return count($matches) >= 1 ? $matches[1] : '';
}
private static function returnR2($str)
{
/*
R2 is the region after the first non-vowel following a vowel in R1, or is the null region at the end
of the word if there is no such non-vowel. Example:
beautiful [ul] beauty [NULL] beau [NULL] animadversion [adversion] sprinkled [NULL] eucharist [ist]
*/
$R1 = self::returnR1($str);
$pattern = '/['.implode(self::$vocali).']+'.'['.implode(self::$consonanti).']'.'(.*)/';
preg_match($pattern, $R1, $matches);
return count($matches) >= 1 ? $matches[1] : '';
}
private static function step0($str)
{
//Step 0: Attached pronoun
//Always do steps 0
$str_len = mb_strlen($str);
$rv = self::returnRV($str);
$rv_len = mb_strlen($rv);
$pos = 0;
foreach (self::$suffissi_step0 as $suff) {
if ($rv_len - mb_strlen($suff) < 0) {
continue;
}
$pos = mb_strpos($rv, $suff, $rv_len - mb_strlen($suff));
if ($pos !== false) {
break;
}
}
$ante_suff = mb_substr($rv, 0, $pos);
$ante_suff_len = mb_strlen($ante_suff);
foreach (self::$ante_suff_a as $ante_a) {
if ($ante_suff_len - mb_strlen($ante_a) < 0) {
continue;
}
$pos_a = mb_strpos($ante_suff, $ante_a, $ante_suff_len - mb_strlen($ante_a));
if ($pos_a !== false) {
return mb_substr($str, 0, $pos + $str_len - $rv_len);
}
}
foreach (self::$ante_suff_b as $ante_b) {
if ($ante_suff_len - mb_strlen($ante_b) < 0) {
continue;
}
$pos_b = mb_strpos($ante_suff, $ante_b, $ante_suff_len - mb_strlen($ante_b));
if ($pos_b !== false) {
return mb_substr($str, 0, $pos + $str_len - $rv_len).'e';
}
}
return $str;
}
private static function deleteStuff($arr_suff, $str, $str_len, $where, $ovunque = false)
{
if ($where === 'r2') {
$r = self::returnR2($str);
} elseif ($where === 'rv') {
$r = self::returnRV($str);
} elseif ($where === 'r1') {
$r = self::returnR1($str);
}
$r_len = mb_strlen($r);
if ($ovunque) {
foreach ($arr_suff as $suff) {
if ($str_len - mb_strlen($suff) < 0) {
continue;
}
$pos = mb_strpos($str, $suff, $str_len - mb_strlen($suff));
if ($pos !== false) {
$pattern = '/'.$suff.'$/';
$ret_str = preg_match($pattern, $r) ? mb_substr($str, 0, $pos) : '';
if ($ret_str !== '') {
return $ret_str;
}
break;
}
}
} else {
foreach ($arr_suff as $suff) {
if ($r_len - mb_strlen($suff) < 0) {
continue;
}
$pos = mb_strpos($r, $suff, $r_len - mb_strlen($suff));
if ($pos !== false) {
return mb_substr($str, 0, $pos + $str_len - $r_len);
}
}
}
}
private static function step1($str)
{
// Step 1: Standard suffix removal
// Always do steps 1
$str_len = mb_strlen($str);
// Delete if in R1, if preceded by 'iv', delete if in R2 (and if further preceded by 'at', delete if in R2),
// otherwise, if preceded by 'os', 'ic' or 'abil', delete if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_g, $str, $str_len, 'r1'))) {
if (!empty($ret_str1 = self::deleteStuff(['iv'], $ret_str, mb_strlen($ret_str), 'r2'))) {
if (!empty($ret_str2 = self::deleteStuff(['at'], $ret_str1, mb_strlen($ret_str1), 'r2'))) {
return $ret_str2;
} else {
return $ret_str1;
}
} elseif (!empty(
$ret_str1 = self::deleteStuff(['os', 'ic', 'abil'], $ret_str, mb_strlen($ret_str), 'r2')
)) {
return $ret_str1;
} else {
return $ret_str;
}
}
// Delete if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_a, $str, $str_len, 'r2', true))) {
return $ret_str;
}
// Delete if in R2, if preceded by 'ic', delete if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_b, $str, $str_len, 'r2'))) {
if (!empty($ret_str1 = self::deleteStuff(['ic'], $ret_str, mb_strlen($ret_str), 'r2'))) {
return $ret_str1;
} else {
return $ret_str;
}
}
// Replace with 'log' if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_c, $str, $str_len, 'r2'))) {
return $ret_str.'log';
}
// Replace with 'u' if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_d, $str, $str_len, 'r2'))) {
return $ret_str.'u';
}
// Replace with 'ente' if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_e, $str, $str_len, 'r2'))) {
return $ret_str.'ente';
}
// Delete if in RV
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_f, $str, $str_len, 'rv'))) {
return $ret_str;
}
// Delete if in R2, if preceded by 'abil', 'ic' or 'iv', delete if in R2
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_h, $str, $str_len, 'r2'))) {
if (!empty($ret_str1 = self::deleteStuff(['abil', 'ic', 'iv'], $ret_str, mb_strlen($ret_str), 'r2'))) {
return $ret_str1;
} else {
return $ret_str;
}
}
// Delete if in R2, if preceded by 'at', delete if in R2 (and if further preceded by 'ic', delete if in R2)
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step1_i, $str, $str_len, 'r2'))) {
if (!empty($ret_str1 = self::deleteStuff(['at'], $ret_str, mb_strlen($ret_str), 'r2'))) {
if (!empty($ret_str2 = self::deleteStuff(['ic'], $ret_str1, mb_strlen($ret_str1), 'r2'))) {
return $ret_str2;
} else {
return $ret_str1;
}
} else {
return $ret_str;
}
}
return $str;
}
private static function step2($str, $str_step1)
{
//Step 2: Verb suffixes
//Do step 2 if no ending was removed by step 1
if ($str != $str_step1) {
return $str_step1;
}
$str_len = mb_strlen($str);
if (!empty($ret_str = self::deleteStuff(self::$suffissi_step2, $str, $str_len, 'rv'))) {
return $ret_str;
}
return $str;
}
private static function step3a($str)
{
// Step 3a: Delete a final 'a', 'e', 'i', 'o',' à', 'è', 'ì' or 'ò' if it is in RV,
// and a preceding 'i' if it is in RV ('crocchi' -> 'crocch', 'crocchio' -> 'crocch')
// Always do steps 3a
$vocale_finale = ['a', 'e', 'i', 'o', 'à', 'è', 'ì', 'ò'];
$str_len = mb_strlen($str);
if (!empty($ret_str = self::deleteStuff($vocale_finale, $str, $str_len, 'rv'))) {
if (!empty($ret_str1 = self::deleteStuff(['i'], $ret_str, mb_strlen($ret_str), 'rv'))) {
return $ret_str1;
} else {
return $ret_str;
}
}
return $str;
}
private static function step3b($str)
{
// Step 3b: Replace final 'ch' (or 'gh') with 'c' (or 'g') if in 'RV' ('crocch' -> 'crocc')
// Always do steps 3b
$rv = self::returnRV($str);
$pattern = '/([cg])h$/';
return mb_substr($str, 0, mb_strlen($str) - mb_strlen($rv))
. preg_replace_callback(
$pattern,
function ($matches) {
return $matches[0];
},
$rv
);
}
private static function step4($str)
{
// Step 4: Finally, turn I and U back into lower case
return strtolower($str);
}
}
@@ -0,0 +1,11 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
class NoStemmer implements Stemmer
{
public static function stem($word)
{
return $word;
}
}
@@ -0,0 +1,144 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
*
* @link https://github.com/Tutanchamon/pl_stemmer
* Simple stemmer for polish language based on pl_stemmer by Błażej Kubiński.
*
*/
class PolishStemmer implements Stemmer
{
public static function removeNouns($word)
{
if (strlen($word) > 7 && in_array(mb_substr($word, -5), array("zacja", "zacją", "zacji"))) {
return mb_substr($word, 0, -4);
}
if (strlen($word) > 6 && in_array(mb_substr($word, -4), array("acja", "acji", "acją", "tach", "anie", "enie", "eniu", "aniu"))) {
return mb_substr($word, 0, -4);
}
if (strlen($word) > 6 && (mb_substr($word, -4) == "tyka")) {
return mb_substr($word, 0, -2);
}
if (strlen($word) > 5 && in_array(mb_substr($word, -3), array("ach", "ami", "nia", "niu", "cia", "ciu"))) {
return mb_substr($word, 0, -3);
}
if (strlen($word) > 5 && in_array(mb_substr($word, -3), array("cji", "cja", "cją"))) {
return mb_substr($word, 0, -2);
}
if (strlen($word) > 5 && in_array(mb_substr($word, -2), array("ce", "ta"))) {
return mb_substr($word, 0, -2);
}
return $word;
}
public static function removeDiminutive($word)
{
if (strlen($word) > 6) {
if (in_array(mb_substr($word, -5), array("eczek", "iczek", "iszek", "aszek", "uszek"))) {
return mb_substr($word, 0, -5);
}
if (in_array(mb_substr($word, -4), array("enek", "ejek", "erek"))) {
return mb_substr($word, 0, -2);
}
}
if (strlen($word) > 4) {
if (in_array(mb_substr($word, -2), array("ek", "ak"))) {
return mb_substr($word, 0, -2);
}
}
return $word;
}
public static function removeAdjectiveEnds($word)
{
if (strlen($word) > 7 && (mb_substr($word, 0, 3) == "naj") && in_array(mb_substr($word, -3), array("sze", "szy"))) {
return mb_substr($word, 3, -3);
}
if (strlen($word) > 7 && (mb_substr($word, 0, 3) == "naj") && (mb_substr($word, 0, 5) == "szych")) {
return mb_substr($word, 3, -5);
}
if (strlen($word) > 6 && (mb_substr($word, -4) == "czny")) {
return mb_substr($word, 0, -4);
}
if (strlen($word) > 5 && in_array(mb_substr($word, -3), array("owy", "owa", "owe", "ych", "ego"))) {
return mb_substr($word, 0, -3);
}
if (strlen($word) > 5 && (mb_substr($word, -2) == "ej")) {
return mb_substr($word, 0, -2);
}
return $word;
}
public static function removeVerbsEnds($word)
{
if (strlen($word) > 5 && (mb_substr($word, -3) == "bym")) {
return mb_substr($word, 0, -3);
}
if (strlen($word) > 5 && in_array(mb_substr($word, -3), array("esz", "asz", "cie", "eść", "aść", "łem", "amy", "emy"))) {
return mb_substr($word, 0, -3);
}
if (strlen($word) > 3 && in_array(mb_substr($word, -3), array("esz", "asz", "eść", "aść", "eć", "ać"))) {
return mb_substr($word, 0, -2);
}
if (strlen($word) > 3 && in_array(mb_substr($word, -2), array("aj"))) {
return mb_substr($word, 0, -1);
}
if (strlen($word) > 3 && in_array(mb_substr($word, -2), array("ać", "em", "am", "ał", "ił", "ić", "ąc"))) {
return mb_substr($word, 0, -2);
}
return $word;
}
public static function removeAdverbsEnds($word)
{
if (strlen($word) > 4 && in_array(mb_substr($word, -3), array("nie", "wie", "rze"))) {
return mb_substr($word, 0, -2);
}
return $word;
}
public static function removePluralForms($word)
{
if (strlen($word) > 4 && in_array(mb_substr($word, -2), array("ów", "om"))) {
return mb_substr($word, 0, -2);
}
if (strlen($word) > 4 && (mb_substr($word, -3) == "ami")) {
return mb_substr($word, 0, -3);
}
return $word;
}
public static function removeGeneralEnds($word)
{
if (strlen($word) > 4 && in_array(substr($word, -2), array("ia", "ie"))) {
return substr($word, 0, -2);
}
if (strlen($word) > 4 && in_array(substr($word, -1), array("u", "ą", "i", "a", "ę", "y", "ę", "ł"))) {
return substr($word, 0, -1);
}
return $word;
}
public static function stem($word)
{
$word = mb_strtolower($word);
$stem = $word;
$stem = self::removeNouns($stem);
$stem = self::removeDiminutive($stem);
$stem = self::removeAdjectiveEnds($stem);
$stem = self::removeVerbsEnds($stem);
$stem = self::removeAdverbsEnds($stem);
$stem = self::removePluralForms($stem);
$stem = self::removeGeneralEnds($stem);
return $stem;
}
}
@@ -0,0 +1,424 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
* Copyright (c) 2005 Richard Heyes (http://www.phpguru.org/)
*
* All rights reserved.
*
* This script is free software.
*/
/**
* PHP5 Implementation of the Porter Stemmer algorithm. Certain elements
* were borrowed from the (broken) implementation by Jon Abernathy.
*
* Usage:
*
* $stem = PorterStemmer::Stem($word);
*
* How easy is that?
*/
class PorterStemmer implements Stemmer
{
/**
* Regex for matching a consonant
* @var string
*/
private static $regex_consonant = '(?:[bcdfghjklmnpqrstvwxz]|(?<=[aeiou])y|^y)';
/**
* Regex for matching a vowel
* @var string
*/
private static $regex_vowel = '(?:[aeiou]|(?<![aeiou])y)';
/**
* Stems a word. Simple huh?
*
* @param string $word Word to stem
* @return string Stemmed word
*/
public static function stem($word)
{
if (strlen($word) <= 2) {
return $word;
}
$word = self::step1ab($word);
$word = self::step1c($word);
$word = self::step2($word);
$word = self::step3($word);
$word = self::step4($word);
$word = self::step5($word);
return $word;
}
/**
* Step 1
* @param string $word
* @return string
*/
private static function step1ab($word)
{
$word = self::doPartA($word);
$word = self::doPartB($word);
return $word;
}
/**
* @param string $word
*/
private static function doPartA($word)
{
if (substr($word, -1) == 's') {
self::replace($word, 'sses', 'ss')
|| self::replace($word, 'ies', 'i')
|| self::replace($word, 'ss', 'ss')
|| self::replace($word, 's', '');
}
return $word;
}
private static function doPartB($word)
{
if (substr($word, -2, 1) != 'e' || !self::replace($word, 'eed', 'ee', 0)) {
// First rule
$v = self::$regex_vowel;
// ing and ed
if (preg_match("#$v+#", substr($word, 0, -3)) && self::replace($word, 'ing', '')
|| preg_match("#$v+#", substr($word, 0, -2)) && self::replace($word, 'ed', '')) {
// Note use of && and OR, for precedence reasons
// If one of above two test successful
if (!self::replace($word, 'at', 'ate')
&& !self::replace($word, 'bl', 'ble')
&& !self::replace($word, 'iz', 'ize')) {
// Double consonant ending
if (self::doubleConsonant($word)
&& substr($word, -2) != 'll'
&& substr($word, -2) != 'ss'
&& substr($word, -2) != 'zz') {
$word = substr($word, 0, -1);
} else if (self::m($word) == 1 && self::cvc($word)) {
$word .= 'e';
}
}
}
}
return $word;
}
/**
* Step 1c
*
* @param string $word Word to stem
*/
private static function step1c($word)
{
$v = self::$regex_vowel;
if (substr($word, -1) == 'y' && preg_match("#$v+#", substr($word, 0, -1))) {
self::replace($word, 'y', 'i');
}
return $word;
}
/**
* Step 2
*
* @param string $word Word to stem
*/
private static function step2($word)
{
switch (substr($word, -2, 1)) {
case 'a':
self::replace($word, 'ational', 'ate', 0)
|| self::replace($word, 'tional', 'tion', 0);
break;
case 'c':
self::replace($word, 'enci', 'ence', 0)
|| self::replace($word, 'anci', 'ance', 0);
break;
case 'e':
self::replace($word, 'izer', 'ize', 0);
break;
case 'g':
self::replace($word, 'logi', 'log', 0);
break;
case 'l':
self::replace($word, 'entli', 'ent', 0)
|| self::replace($word, 'ousli', 'ous', 0)
|| self::replace($word, 'alli', 'al', 0)
|| self::replace($word, 'bli', 'ble', 0)
|| self::replace($word, 'eli', 'e', 0);
break;
case 'o':
self::replace($word, 'ization', 'ize', 0)
|| self::replace($word, 'ation', 'ate', 0)
|| self::replace($word, 'ator', 'ate', 0);
break;
case 's':
self::replace($word, 'iveness', 'ive', 0)
|| self::replace($word, 'fulness', 'ful', 0)
|| self::replace($word, 'ousness', 'ous', 0)
|| self::replace($word, 'alism', 'al', 0);
break;
case 't':
self::replace($word, 'biliti', 'ble', 0)
|| self::replace($word, 'aliti', 'al', 0)
|| self::replace($word, 'iviti', 'ive', 0);
break;
}
return $word;
}
/**
* Step 3
*
* @param string $word String to stem
*/
private static function step3($word)
{
switch (substr($word, -2, 1)) {
case 'a':
self::replace($word, 'ical', 'ic', 0);
break;
case 's':
self::replace($word, 'ness', '', 0);
break;
case 't':
self::replace($word, 'icate', 'ic', 0)
|| self::replace($word, 'iciti', 'ic', 0);
break;
case 'u':
self::replace($word, 'ful', '', 0);
break;
case 'v':
self::replace($word, 'ative', '', 0);
break;
case 'z':
self::replace($word, 'alize', 'al', 0);
break;
}
return $word;
}
/**
* Step 4
*
* @param string $word Word to stem
*/
private static function step4($word)
{
switch (substr($word, -2, 1)) {
case 'a':
self::replace($word, 'al', '', 1);
break;
case 'c':
self::replace($word, 'ance', '', 1)
|| self::replace($word, 'ence', '', 1);
break;
case 'e':
self::replace($word, 'er', '', 1);
break;
case 'i':
self::replace($word, 'ic', '', 1);
break;
case 'l':
self::replace($word, 'able', '', 1)
|| self::replace($word, 'ible', '', 1);
break;
case 'n':
self::replace($word, 'ant', '', 1)
|| self::replace($word, 'ement', '', 1)
|| self::replace($word, 'ment', '', 1)
|| self::replace($word, 'ent', '', 1);
break;
case 'o':
if (substr($word, -4) == 'tion' || substr($word, -4) == 'sion') {
self::replace($word, 'ion', '', 1);
} else {
self::replace($word, 'ou', '', 1);
}
break;
case 's':
self::replace($word, 'ism', '', 1);
break;
case 't':
self::replace($word, 'ate', '', 1)
|| self::replace($word, 'iti', '', 1);
break;
case 'u':
self::replace($word, 'ous', '', 1);
break;
case 'v':
self::replace($word, 'ive', '', 1);
break;
case 'z':
self::replace($word, 'ize', '', 1);
break;
}
return $word;
}
/**
* Step 5
*
* @param string $word Word to stem
*/
private static function step5($word)
{
// Part a
if (substr($word, -1) == 'e') {
if (self::m(substr($word, 0, -1)) > 1) {
self::replace($word, 'e', '');
} else if (self::m(substr($word, 0, -1)) == 1) {
if (!self::cvc(substr($word, 0, -1))) {
self::replace($word, 'e', '');
}
}
}
// Part b
if (self::m($word) > 1 && self::doubleConsonant($word) && substr($word, -1) == 'l') {
$word = substr($word, 0, -1);
}
return $word;
}
/**
* Replaces the first string with the second, at the end of the string. If third
* arg is given, then the preceding string must match that m count at least.
*
* @param string $str String to check
* @param string $check Ending to check for
* @param string $repl Replacement string
* @param int $m Optional minimum number of m() to meet
* @return bool Whether the $check string was at the end
* of the $str string. True does not necessarily mean
* that it was replaced.
*/
private static function replace(&$str, $check, $repl, $m = null)
{
$len = 0 - strlen($check);
if (substr($str, $len) == $check) {
$substr = substr($str, 0, $len);
if (is_null($m) || self::m($substr) > $m) {
$str = $substr.$repl;
}
return true;
}
return false;
}
/**
* What, you mean it's not obvious from the name?
*
* m() measures the number of consonant sequences in $str. if c is
* a consonant sequence and v a vowel sequence, and <..> indicates arbitrary
* presence,
*
* <c><v> gives 0
* <c>vc<v> gives 1
* <c>vcvc<v> gives 2
* <c>vcvcvc<v> gives 3
*
* @param string $str The string to return the m count for
* @return int The m count
*/
private static function m($str)
{
$c = self::$regex_consonant;
$v = self::$regex_vowel;
$str = preg_replace("#^$c+#", '', $str);
$str = preg_replace("#$v+$#", '', $str);
preg_match_all("#($v+$c+)#", $str, $matches);
return count($matches[1]);
}
/**
* Returns true/false as to whether the given string contains two
* of the same consonant next to each other at the end of the string.
*
* @param string $str String to check
* @return bool Result
*/
private static function doubleConsonant($str)
{
$c = self::$regex_consonant;
return preg_match("#$c{2}$#", $str, $matches) && $matches[0][0] == $matches[0][1];
}
/**
* Checks for ending CVC sequence where second C is not W, X or Y
*
* @param string $str String to check
* @return bool Result
*/
private static function cvc($str)
{
$c = self::$regex_consonant;
$v = self::$regex_vowel;
$matchFound = preg_match("#($c$v$c)$#", $str, $matches);
$return = false;
if ($matchFound && strlen($matches[1]) == 3) {
$return = true;
if (in_array($matches[1][2], ['w', 'x', 'y'])) {
$return = false;
}
}
return $return;
}
}
@@ -0,0 +1,727 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
* This program is free software: you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation, either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program. If not, see <http://www.gnu.org/licenses/>.
*/
/**
* This is a reimplementation of the Porter Stemmer Algorithm for Portuguese.
* This script is based on the implementation found on <https://github.com/wamania/php-stemmer>
* and has been rewriten to work with TNTSearch by Lucas Padilha <https://github.com/LucasPadilha>
*
* Takes a word and reduces it to its Portuguese stem using the Porter stemmer algorithm.
*
* References:
* - http://snowball.tartarus.org/algorithms/porter/stemmer.html
* - http://snowball.tartarus.org/algorithms/portuguese/stemmer.html
*
* Usage:
* $stem = PortugueseStemmer::stem($word);
*
* @author Lucas Padilha <https://github.com/LucasPadilha>
*/
class PortugueseStemmer implements Stemmer
{
/**
* UTF-8 Case lookup table
*
* This lookuptable defines the upper case letters to their correspponding
* lower case letter in UTF-8
*
* @author Andreas Gohr <andi@splitbrain.org>
*/
private static $utf8_lower_to_upper = array(
0x0061=>0x0041, 0x03C6=>0x03A6, 0x0163=>0x0162, 0x00E5=>0x00C5, 0x0062=>0x0042,
0x013A=>0x0139, 0x00E1=>0x00C1, 0x0142=>0x0141, 0x03CD=>0x038E, 0x0101=>0x0100,
0x0491=>0x0490, 0x03B4=>0x0394, 0x015B=>0x015A, 0x0064=>0x0044, 0x03B3=>0x0393,
0x00F4=>0x00D4, 0x044A=>0x042A, 0x0439=>0x0419, 0x0113=>0x0112, 0x043C=>0x041C,
0x015F=>0x015E, 0x0144=>0x0143, 0x00EE=>0x00CE, 0x045E=>0x040E, 0x044F=>0x042F,
0x03BA=>0x039A, 0x0155=>0x0154, 0x0069=>0x0049, 0x0073=>0x0053, 0x1E1F=>0x1E1E,
0x0135=>0x0134, 0x0447=>0x0427, 0x03C0=>0x03A0, 0x0438=>0x0418, 0x00F3=>0x00D3,
0x0440=>0x0420, 0x0454=>0x0404, 0x0435=>0x0415, 0x0449=>0x0429, 0x014B=>0x014A,
0x0431=>0x0411, 0x0459=>0x0409, 0x1E03=>0x1E02, 0x00F6=>0x00D6, 0x00F9=>0x00D9,
0x006E=>0x004E, 0x0451=>0x0401, 0x03C4=>0x03A4, 0x0443=>0x0423, 0x015D=>0x015C,
0x0453=>0x0403, 0x03C8=>0x03A8, 0x0159=>0x0158, 0x0067=>0x0047, 0x00E4=>0x00C4,
0x03AC=>0x0386, 0x03AE=>0x0389, 0x0167=>0x0166, 0x03BE=>0x039E, 0x0165=>0x0164,
0x0117=>0x0116, 0x0109=>0x0108, 0x0076=>0x0056, 0x00FE=>0x00DE, 0x0157=>0x0156,
0x00FA=>0x00DA, 0x1E61=>0x1E60, 0x1E83=>0x1E82, 0x00E2=>0x00C2, 0x0119=>0x0118,
0x0146=>0x0145, 0x0070=>0x0050, 0x0151=>0x0150, 0x044E=>0x042E, 0x0129=>0x0128,
0x03C7=>0x03A7, 0x013E=>0x013D, 0x0442=>0x0422, 0x007A=>0x005A, 0x0448=>0x0428,
0x03C1=>0x03A1, 0x1E81=>0x1E80, 0x016D=>0x016C, 0x00F5=>0x00D5, 0x0075=>0x0055,
0x0177=>0x0176, 0x00FC=>0x00DC, 0x1E57=>0x1E56, 0x03C3=>0x03A3, 0x043A=>0x041A,
0x006D=>0x004D, 0x016B=>0x016A, 0x0171=>0x0170, 0x0444=>0x0424, 0x00EC=>0x00CC,
0x0169=>0x0168, 0x03BF=>0x039F, 0x006B=>0x004B, 0x00F2=>0x00D2, 0x00E0=>0x00C0,
0x0434=>0x0414, 0x03C9=>0x03A9, 0x1E6B=>0x1E6A, 0x00E3=>0x00C3, 0x044D=>0x042D,
0x0436=>0x0416, 0x01A1=>0x01A0, 0x010D=>0x010C, 0x011D=>0x011C, 0x00F0=>0x00D0,
0x013C=>0x013B, 0x045F=>0x040F, 0x045A=>0x040A, 0x00E8=>0x00C8, 0x03C5=>0x03A5,
0x0066=>0x0046, 0x00FD=>0x00DD, 0x0063=>0x0043, 0x021B=>0x021A, 0x00EA=>0x00CA,
0x03B9=>0x0399, 0x017A=>0x0179, 0x00EF=>0x00CF, 0x01B0=>0x01AF, 0x0065=>0x0045,
0x03BB=>0x039B, 0x03B8=>0x0398, 0x03BC=>0x039C, 0x045C=>0x040C, 0x043F=>0x041F,
0x044C=>0x042C, 0x00FE=>0x00DE, 0x00F0=>0x00D0, 0x1EF3=>0x1EF2, 0x0068=>0x0048,
0x00EB=>0x00CB, 0x0111=>0x0110, 0x0433=>0x0413, 0x012F=>0x012E, 0x00E6=>0x00C6,
0x0078=>0x0058, 0x0161=>0x0160, 0x016F=>0x016E, 0x03B1=>0x0391, 0x0457=>0x0407,
0x0173=>0x0172, 0x00FF=>0x0178, 0x006F=>0x004F, 0x043B=>0x041B, 0x03B5=>0x0395,
0x0445=>0x0425, 0x0121=>0x0120, 0x017E=>0x017D, 0x017C=>0x017B, 0x03B6=>0x0396,
0x03B2=>0x0392, 0x03AD=>0x0388, 0x1E85=>0x1E84, 0x0175=>0x0174, 0x0071=>0x0051,
0x0437=>0x0417, 0x1E0B=>0x1E0A, 0x0148=>0x0147, 0x0105=>0x0104, 0x0458=>0x0408,
0x014D=>0x014C, 0x00ED=>0x00CD, 0x0079=>0x0059, 0x010B=>0x010A, 0x03CE=>0x038F,
0x0072=>0x0052, 0x0430=>0x0410, 0x0455=>0x0405, 0x0452=>0x0402, 0x0127=>0x0126,
0x0137=>0x0136, 0x012B=>0x012A, 0x03AF=>0x038A, 0x044B=>0x042B, 0x006C=>0x004C,
0x03B7=>0x0397, 0x0125=>0x0124, 0x0219=>0x0218, 0x00FB=>0x00DB, 0x011F=>0x011E,
0x043E=>0x041E, 0x1E41=>0x1E40, 0x03BD=>0x039D, 0x0107=>0x0106, 0x03CB=>0x03AB,
0x0446=>0x0426, 0x00FE=>0x00DE, 0x00E7=>0x00C7, 0x03CA=>0x03AA, 0x0441=>0x0421,
0x0432=>0x0412, 0x010F=>0x010E, 0x00F8=>0x00D8, 0x0077=>0x0057, 0x011B=>0x011A,
0x0074=>0x0054, 0x006A=>0x004A, 0x045B=>0x040B, 0x0456=>0x0406, 0x0103=>0x0102,
0x03BB=>0x039B, 0x00F1=>0x00D1, 0x043D=>0x041D, 0x03CC=>0x038C, 0x00E9=>0x00C9,
0x00F0=>0x00D0, 0x0457=>0x0407, 0x0123=>0x0122,
);
private static $vowels = array('a', 'e', 'i', 'o', 'u', 'á', 'é', 'í', 'ó', 'ú', 'â', 'ê', 'ô');
public static function stem($word)
{
// we do ALL in UTF-8
if (!self::check($word)) {
throw new \Exception('Word must be in UTF-8');
}
$word = self::strtolower($word);
$word = self::str_replace(array('ã', 'õ'), array('a~', 'o~'), $word);
$rv = '';
$rvIndex = '';
self::rv($word, $rv, $rvIndex);
$r1 = '';
$r1Index = '';
self::r1($word, $r1, $r1Index);
$r2 = '';
$r2Index = '';
self::r2($r1, $r1Index, $r2, $r2Index);
$initialWord = $word;
self::step1($word, $r1Index, $r2Index, $rvIndex);
if ($initialWord == $word) {
self::step2($word, $rvIndex);
}
if ($initialWord != $word) {
self::step3($word, $rvIndex);
} else {
self::step4($word, $rvIndex);
}
self::step5($word, $rvIndex);
self::finish($word);
return $word;
}
/**
* R1 is the region after the first non-vowel following a vowel, or the end of the word if there is no such non-vowel.
*/
private static function r1($word, &$r1, &$r1Index)
{
list($index, $value) = self::rx($word);
$r1 = $value;
$r1Index = $index;
return true;
}
/**
* R2 is the region after the first non-vowel following a vowel in R1, or the end of the word if there is no such non-vowel.
*/
private static function r2($r1, $r1Index, &$r2, &$r2Index)
{
list($index, $value) = self::rx($r1);
$r2 = $value;
$r2Index = $r1Index + $index;
return true;
}
/**
* Common function for R1 and R2
* Search the region after the first non-vowel following a vowel in $word, or the end of the word if there is no such non-vowel.
* R1 : $in = $this->word
* R2 : $in = R1
*/
private static function rx($in)
{
$length = self::strlen($in);
// Defaults
$value = '';
$index = $length;
// Search all vowels
$vowels = array();
for ($i = 0; $i < $length; $i++) {
$letter = self::substr($in, $i, 1);
if (in_array($letter, static::$vowels)) {
$vowels[] = $i;
}
}
// Search the non-vowel following a vowel
foreach ($vowels as $position) {
$after = $position + 1;
$letter = self::substr($in, $after, 1);
if (!in_array($letter, static::$vowels)) {
$index = $after + 1;
$value = self::substr($in, ($after+1));
break;
}
}
return array($index, $value);
}
/**
* Used by spanish, italian, portuguese, etc (but not by french)
*
* If the second letter is a consonant, RV is the region after the next following vowel,
* or if the first two letters are vowels, RV is the region after the next consonant,
* and otherwise (consonant-vowel case) RV is the region after the third letter.
* But RV is the end of the word if these positions cannot be found.
*/
private static function rv($word, &$rv, &$rvIndex)
{
$length = self::strlen($word);
if ($length < 3) {
return true;
}
$first = self::substr($word, 0, 1);
$second = self::substr($word, 1, 1);
// If the second letter is a consonant, RV is the region after the next following vowel,
if (!in_array($second, static::$vowels)) {
for ($i = 2; $i < $length; $i++) {
$letter = self::substr($word, $i, 1);
if (in_array($letter, static::$vowels)) {
$rv = self::substr($word, ($i + 1));
$rvIndex = $i + 1;
return true;
}
}
}
// or if the first two letters are vowels, RV is the region after the next consonant,
if ((in_array($first, static::$vowels)) && (in_array($second, static::$vowels))) {
for ($i = 2; $i < $length; $i++) {
$letter = self::substr($word, $i, 1);
if (!in_array($letter, static::$vowels)) {
$rv = self::substr($word, ($i + 1));
$rvIndex = $i + 1;
return true;
}
}
}
// and otherwise (consonant-vowel case) RV is the region after the third letter.
if ((!in_array($first, static::$vowels)) && (in_array($second, static::$vowels))) {
$rv = self::substr($word, 3);
$rvIndex = 3;
return true;
}
return false;
}
private static function inRv($position, $rvIndex)
{
return ($position >= $rvIndex);
}
private static function inR1($position, $r1Index)
{
return ($position >= $r1Index);
}
private static function inR2($position, $r2Index)
{
return ($position >= $r2Index);
}
private static function searchIfInRv($word, $suffixes, $rvIndex)
{
return self::search($word, $suffixes, $rvIndex);
}
private static function searchIfInR2($word, $suffixes, $r2Index)
{
return self::search($word, $suffixes, $r2Index);
}
private static function search($word, $suffixes, $offset = 0)
{
$length = self::strlen($word);
if ($offset > $length) {
return false;
}
foreach ($suffixes as $suffix) {
if ((($position = self::strrpos($word, $suffix, $offset)) !== false) && ((self::strlen($suffix) + $position) == $length)) {
return $position;
}
}
return false;
}
/**
* Step 1: Standard suffix removal
*/
private static function step1(&$word, $r1Index, $r2Index, $rvIndex)
{
// delete if in R2
if (($position = self::search($word, array('amentos', 'imentos', 'adoras', 'adores', 'amento', 'imento', 'adora', 'istas', 'ismos', 'antes', 'ância', 'ezas', 'eza', 'icos', 'icas', 'ismo', 'ável', 'ível', 'ista', 'oso', 'osos', 'osas', 'osa', 'ico', 'ica', 'ador', 'aça~o', 'aço~es' , 'ante'))) !== false) {
if (self::inR2($position, $r2Index)) {
$word = self::substr($word, 0, $position);
}
return true;
}
// replace with log if in R2
if (($position = self::search($word, array('logías', 'logía'))) !== false) {
if (self::inR2($position, $r2Index)) {
$word = preg_replace('#(logías|logía)$#u', 'log', $word);
}
return true;
}
// replace with u if in R2
if (($position = self::search($word, array('uciones', 'ución'))) !== false) {
if (self::inR2($position, $r2Index)) {
$word = preg_replace('#(uciones|ución)$#u', 'u', $word);
}
return true;
}
// replace with ente if in R2
if (($position = self::search($word, array('ências', 'ência'))) !== false) {
if (self::inR2($position, $r2Index)) {
$word = preg_replace('#(ências|ência)$#u', 'ente', $word);
}
return true;
}
// delete if in R1
// if preceded by iv, delete if in R2 (and if further preceded by at, delete if in R2), otherwise,
// if preceded by os, ic or ad, delete if in R2
if (($position = self::search($word, array('amente'))) !== false) {
// delete if in R1
if (self::inR1($position, $r1Index)) {
$word = self::substr($word, 0, $position);
}
// if preceded by iv, delete if in R2 (and if further preceded by at, delete if in R2), otherwise,
if (($position2 = self::searchIfInR2($word, array('iv'), $r2Index)) !== false) {
$word = self::substr($word, 0, $position2);
if (($position3 = self::searchIfInR2($word, array('at'), $r2Index)) !== false) {
$word = self::substr($word, 0, $position3);
}
// if preceded by os, ic or ad, delete if in R2
} elseif (($position4 = self::searchIfInR2($word, array('os', 'ic', 'ad'), $r2Index)) !== false) {
$word = self::substr($word, 0, $position4);
}
return true;
}
// delete if in R2
// if preceded by ante, avel or ível, delete if in R2
if (($position = self::search($word, array('mente'))) !== false) {
// delete if in R2
if (self::inR2($position, $r2Index)) {
$word = self::substr($word, 0, $position);
}
// if preceded by ante, avel or ível, delete if in R2
if (($position2 = self::searchIfInR2($word, array('ante', 'avel', 'ível'), $r2Index)) != false) {
$word = self::substr($word, 0, $position2);
}
return true;
}
// delete if in R2
// if preceded by abil, ic or iv, delete if in R2
if (($position = self::search($word, array('idades', 'idade'))) !== false) {
// delete if in R2
if (self::inR2($position, $r2Index)) {
$word = self::substr($word, 0, $position);
}
// if preceded by abil, ic or iv, delete if in R2
if (($position2 = self::searchIfInR2($word, array('abil', 'ic', 'iv'), $r2Index)) !== false) {
$word = self::substr($word, 0, $position2);
}
return true;
}
// delete if in R2
// if preceded by at, delete if in R2
if (($position = self::search($word, array('ivas', 'ivos', 'iva', 'ivo'))) !== false) {
// delete if in R2
if (self::inR2($position, $r2Index)) {
$word = self::substr($word, 0, $position);
}
// if preceded by at, delete if in R2
if (($position2 = self::searchIfInR2($word, array('at'), $r2Index)) !== false) {
$word = self::substr($word, 0, $position2);
}
return true;
}
// replace with ir if in RV and preceded by e
if (($position = self::search($word, array('iras', 'ira'))) !== false) {
if (self::inRv($position, $rvIndex)) {
$before = $position - 1;
$letter = self::substr($word, $before, 1);
if ($letter == 'e') {
$word = preg_replace('#(iras|ira)$#u', 'ir', $word);
}
}
return true;
}
return false;
}
/**
* Step 2: Verb suffixes
* Search for the longest among the following suffixes in RV, and if found, delete.
*/
private static function step2(&$word, $rvIndex)
{
if (($position = self::searchIfInRv($word, array('aríamos', 'eríamos', 'iríamos', 'ássemos', 'êssemos', 'íssemos', 'aríeis', 'eríeis', 'iríeis', 'ásseis', 'ésseis', 'ísseis', 'áramos', 'éramos', 'íramos', 'ávamos', 'aremos', 'eremos', 'iremos', 'ariam', 'eriam', 'iriam', 'assem', 'essem', 'issem', 'arias', 'erias', 'irias', 'ardes', 'erdes', 'irdes', 'asses', 'esses', 'isses', 'astes', 'estes', 'istes', 'áreis', 'areis', 'éreis', 'ereis', 'íreis', 'ireis', 'áveis', 'íamos', 'armos', 'ermos', 'irmos', 'aria', 'eria', 'iria', 'asse', 'esse', 'isse', 'aste', 'este', 'iste', 'arei', 'erei', 'irei', 'adas', 'idas', 'aram', 'eram', 'iram', 'avam', 'arem', 'erem', 'irem', 'ando', 'endo', 'indo', 'ara~o', 'era~o', 'ira~o', 'arás', 'aras', 'erás', 'eras', 'irás', 'avas', 'ares', 'eres', 'ires', 'íeis', 'ados', 'idos', 'ámos', 'amos', 'emos', 'imos', 'iras', 'ada', 'ida', 'ará', 'ara', 'erá', 'era', 'irá', 'ava', 'iam', 'ado', 'ido', 'ias', 'ais', 'eis', 'ira', 'ia', 'ei', 'am', 'em', 'ar', 'er', 'ir', 'as', 'es', 'is', 'eu', 'iu', 'ou'), $rvIndex)) !== false) {
$word = self::substr($word, 0, $position);
return true;
}
return false;
}
/**
* Step 3: d-suffixes
*
*/
private static function step3(&$word, $rvIndex)
{
// Delete suffix i if in RV and preceded by c
if (self::searchIfInRv($word, array('i'), $rvIndex) !== false) {
$letter = self::substr($word, -2, 1);
if ($letter == 'c') {
$word = self::substr($word, 0, -1);
}
return true;
}
return false;
}
/**
* Step 4
*/
private static function step4(&$word, $rvIndex)
{
// If the word ends with one of the suffixes "os a i o á í ó" in RV, delete it
if (($position = self::searchIfInRv($word, array('os', 'a', 'i', 'o','á', 'í', 'ó'), $rvIndex)) !== false) {
$word = self::substr($word, 0, $position);
return true;
}
return false;
}
/**
* Step 5
*/
private static function step5(&$word, $rvIndex)
{
// If the word ends with one of "e é ê" in RV, delete it, and if preceded by gu (or ci) with the u (or i) in RV, delete the u (or i).
if (self::searchIfInRv($word, array('e', 'é', 'ê'), $rvIndex) !== false) {
$word = self::substr($word, 0, -1);
if (($position2 = self::search($word, array('gu', 'ci'))) !== false) {
if (self::inRv(($position2 + 1), $rvIndex)) {
$word = self::substr($word, 0, -1);
}
}
return true;
} elseif (self::search($word, array('ç')) !== false) {
$word = preg_replace('#(ç)$#u', 'c', $word);
return true;
}
return false;
}
private static function finish(&$word)
{
// turn U and Y back into lower case, and remove the umlaut accent from a, o and u.
$word = self::str_replace(array('a~', 'o~'), array('ã', 'õ'), $word);
}
/**
* Tries to detect if a string is in Unicode encoding
*
* @author <bmorel@ssi.fr>
* @link http://www.php.net/manual/en/function.utf8-encode.php
*/
private static function check($str)
{
for ($i=0; $i<strlen($str); $i++) {
if (ord($str[$i]) < 0x80) continue; # 0bbbbbbb
elseif ((ord($str[$i]) & 0xE0) == 0xC0) $n=1; # 110bbbbb
elseif ((ord($str[$i]) & 0xF0) == 0xE0) $n=2; # 1110bbbb
elseif ((ord($str[$i]) & 0xF8) == 0xF0) $n=3; # 11110bbb
elseif ((ord($str[$i]) & 0xFC) == 0xF8) $n=4; # 111110bb
elseif ((ord($str[$i]) & 0xFE) == 0xFC) $n=5; # 1111110b
else return false; # Does not match any model
for ($j=0; $j<$n; $j++) { # n bytes matching 10bbbbbb follow ?
if ((++$i == strlen($str)) || ((ord($str[$i]) & 0xC0) != 0x80))
return false;
}
}
return true;
}
/**
* Unicode aware replacement for strlen()
*
* utf8_decode() converts characters that are not in ISO-8859-1
* to '?', which, for the purpose of counting, is alright - It's
* even faster than mb_strlen.
*
* @author <chernyshevsky at hotmail dot com>
* @see strlen()
* @see utf8_decode()
*/
private static function strlen($string)
{
return strlen(utf8_decode($string));
}
/**
* Unicode aware replacement for substr()
*
* @author lmak at NOSPAM dot iti dot gr
* @link http://www.php.net/manual/en/function.substr.php
* @see substr()
*/
private static function substr($str,$start,$length=null)
{
$ar = array();
preg_match_all("/./u", $str, $ar);
if($length != null) {
return join("",array_slice($ar[0],$start,$length));
} else {
return join("",array_slice($ar[0],$start));
}
}
/**
* Unicode aware replacement for strrepalce()
*
* @author Harry Fuecks <hfuecks@gmail.com>
* @see strreplace();
*/
private static function str_replace($s,$r,$str)
{
if(!is_array($s)){
$s = '!'.preg_quote($s,'!').'!u';
}else{
foreach ($s as $k => $v) {
$s[$k] = '!'.preg_quote($v).'!u';
}
}
return preg_replace($s,$r,$str);
}
/**
* This is a unicode aware replacement for strtolower()
*
* Uses mb_string extension if available
*
* @author Andreas Gohr <andi@splitbrain.org>
* @see strtolower()
* @see utf8_strtoupper()
*/
private static function strtolower($string)
{
if(!defined('UTF8_NOMBSTRING') && function_exists('mb_strtolower'))
return mb_strtolower($string,'utf-8');
//global $utf8_upper_to_lower;
$utf8_upper_to_lower = array_flip(self::$utf8_lower_to_upper);
$uni = self::utf8_to_unicode($string);
$cnt = count($uni);
for ($i=0; $i < $cnt; $i++){
if($utf8_upper_to_lower[$uni[$i]]){
$uni[$i] = $utf8_upper_to_lower[$uni[$i]];
}
}
return self::unicode_to_utf8($uni);
}
/**
* This function returns any UTF-8 encoded text as a list of
* Unicode values:
*
* @author Scott Michael Reynen <scott@randomchaos.com>
* @link http://www.randomchaos.com/document.php?source=php_and_unicode
* @see unicode_to_utf8()
*/
private static function utf8_to_unicode( &$str )
{
$unicode = array();
$values = array();
$looking_for = 1;
for ($i = 0; $i < strlen( $str ); $i++ ) {
$this_value = ord( $str[ $i ] );
if ( $this_value < 128 ) $unicode[] = $this_value;
else {
if ( count( $values ) == 0 ) $looking_for = ( $this_value < 224 ) ? 2 : 3;
$values[] = $this_value;
if ( count( $values ) == $looking_for ) {
$number = ( $looking_for == 3 ) ?
( ( $values[0] % 16 ) * 4096 ) + ( ( $values[1] % 64 ) * 64 ) + ( $values[2] % 64 ):
( ( $values[0] % 32 ) * 64 ) + ( $values[1] % 64 );
$unicode[] = $number;
$values = array();
$looking_for = 1;
}
}
}
return $unicode;
}
/**
* This function converts a Unicode array back to its UTF-8 representation
*
* @author Scott Michael Reynen <scott@randomchaos.com>
* @link http://www.randomchaos.com/document.php?source=php_and_unicode
* @see utf8_to_unicode()
*/
private static function unicode_to_utf8( &$str )
{
if (!is_array($str)) return '';
$utf8 = '';
foreach( $str as $unicode ) {
if ( $unicode < 128 ) {
$utf8.= chr( $unicode );
} elseif ( $unicode < 2048 ) {
$utf8.= chr( 192 + ( ( $unicode - ( $unicode % 64 ) ) / 64 ) );
$utf8.= chr( 128 + ( $unicode % 64 ) );
} else {
$utf8.= chr( 224 + ( ( $unicode - ( $unicode % 4096 ) ) / 4096 ) );
$utf8.= chr( 128 + ( ( ( $unicode % 4096 ) - ( $unicode % 64 ) ) / 64 ) );
$utf8.= chr( 128 + ( $unicode % 64 ) );
}
}
return $utf8;
}
/**
* This is an Unicode aware replacement for strrpos
*
* Uses mb_string extension if available
*
* @author Harry Fuecks <hfuecks@gmail.com>
* @see strpos()
*/
private static function strrpos($haystack, $needle, $offset=0)
{
if(!defined('UTF8_NOMBSTRING') && function_exists('mb_strrpos'))
return mb_strrpos($haystack, $needle, $offset, 'utf-8');
if (!$offset) {
$ar = self::explode($needle, $haystack);
$count = count($ar);
if ( $count > 1 ) {
return self::strlen($haystack) - self::strlen($ar[($count-1)]) - self::strlen($needle);
}
return false;
} else {
if ( !is_int($offset) ) {
trigger_error('Offset must be an integer', E_USER_WARNING);
return false;
}
$str = self::substr($haystack, $offset);
if ( false !== ($pos = self::strrpos($str, $needle))){
return $pos + $offset;
}
return false;
}
}
/**
* Unicode aware replacement for explode
*
* @author Harry Fuecks <hfuecks@gmail.com>
* @see explode();
*/
private static function explode($sep, $str)
{
if ( $sep == '' ) {
trigger_error('Empty delimiter',E_USER_WARNING);
return FALSE;
}
return preg_split('!'.preg_quote($sep,'!').'!u',$str);
}
}
@@ -0,0 +1,83 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
* Semple stemmer for russian language
*/
class RussianStemmer implements Stemmer
{
private static $VOWEL = '/аеиоуыэюя/u';
private static $PERFECTIVEGROUND = '/((ив|ивши|ившись|ыв|ывши|ывшись)|((?<=[ая])(в|вши|вшись)))$/u';
private static $REFLEXIVE = '/(с[яь])$/u';
private static $ADJECTIVE = '/(ее|ие|ые|ое|ими|ыми|ей|ий|ый|ой|ем|им|ым|ом|его|ого|ему|ому|их|ых|ую|юю|ая|яя|ою|ею)$/u';
private static $PARTICIPLE = '/((ивш|ывш|ующ)|((?<=[ая])(ем|нн|вш|ющ|щ)))$/u';
private static $VERB = '/((ила|ыла|ена|ейте|уйте|ите|или|ыли|ей|уй|ил|ыл|им|ым|ен|ило|ыло|ено|ят|ует|уют|ит|ыт|ены|ить|ыть|ишь|ую|ю)|((?<=[ая])(ла|на|ете|йте|ли|й|л|ем|н|ло|но|ет|ют|ны|ть|ешь|нно)))$/u';
private static $NOUN = '/(а|ев|ов|ие|ье|е|иями|ями|ами|еи|ии|и|ией|ей|ой|ий|й|иям|ям|ием|ем|ам|ом|о|у|ах|иях|ях|ы|ь|ию|ью|ю|ия|ья|я)$/u';
private static $RVRE = '/^(.*?[аеиоуыэюя])(.*)$/u';
private static $DERIVATIONAL = '/[^аеиоуыэюя][аеиоуыэюя]+[^аеиоуыэюя]+[аеиоуыэюя].*(?<=о)сть?$/u';
private static function s(&$s, $re, $to)
{
$orig = $s;
$s = preg_replace($re, $to, $s);
return $orig !== $s;
}
private static function m($s, $re)
{
return preg_match($re, $s);
}
public static function stem($word)
{
$word = mb_strtolower($word);
$word = str_replace('ё', 'е', $word);
$stem = $word;
do {
if (!preg_match(self::$RVRE, $word, $p)) {
break;
}
$start = $p[1];
$RV = $p[2];
if (!$RV) {
break;
}
// Step 1
if (!self::s($RV, self::$PERFECTIVEGROUND, '')) {
self::s($RV, self::$REFLEXIVE, '');
if (self::s($RV, self::$ADJECTIVE, '')) {
self::s($RV, self::$PARTICIPLE, '');
} else {
if (!self::s($RV, self::$VERB, '')) {
self::s($RV, self::$NOUN, '');
}
}
}
// Step 2
self::s($RV, '/и$/u', '');
// Step 3
if (self::m($RV, self::$DERIVATIONAL)) {
self::s($RV, '/ость?$/u', '');
}
// Step 4
if (!self::s($RV, '/ь$/u', '')) {
self::s($RV, '/ейше?/u', '');
self::s($RV, '/нн$/u', 'н');
}
$stem = $start . $RV;
} while (FALSE);
return $stem;
}
}
@@ -0,0 +1,6 @@
<?php namespace TeamTNT\TNTSearch\Stemmer;
interface Stemmer
{
public static function stem($word);
}
@@ -0,0 +1,83 @@
<?php
namespace TeamTNT\TNTSearch\Stemmer;
/**
* Semple stemmer for ukrainian language
*/
class UkrainianStemmer implements Stemmer
{
private static $VOWEL = '/аеиоуюяіїє/u';
/* http://uk.wikipedia.org/wiki/Голосний_звук */
// var $PERFECTIVEGROUND = '/((ив|ивши|ившись|ыв|ывши|ывшись((?<=[ая])(в|вши|вшись)))$/';
private static $PERFECTIVEGROUND = '/(ив|ивши|ившись|ів|івши|івшись((?<=[ая|я])(в|вши|вшись)))$/u';
private static $REFLEXIVE = '/(с[яьи])$/u'; // http://uk.wikipedia.org/wiki/Рефлексивне_дієслово
private static $ADJECTIVE = '/(ими|ій|ий|а|е|ова|ове|ів|є|їй|єє|еє|я|ім|ем|им|ім|их|іх|ою|йми|іми|у|ю|ого|ому|ої)$/u'; //http://uk.wikipedia.org/wiki/Прикметник + http://wapedia.mobi/uk/Прикметник
private static $PARTICIPLE = '/(ий|ого|ому|им|ім|а|ій|у|ою|ій|і|их|йми|их)$/u'; //http://uk.wikipedia.org/wiki/Дієприкметник
private static $VERB = '/(сь|ся|ив|ать|ять|у|ю|ав|али|учи|ячи|вши|ши|е|ме|ати|яти|є)$/u'; //http://uk.wikipedia.org/wiki/Дієслово
private static $NOUN = '/(а|ев|ов|е|ями|ами|еи|и|ей|ой|ий|й|иям|ям|ием|ем|ам|ом|о|у|ах|иях|ях|ы|ь|ию|ью|ю|ия|ья|я|і|ові|ї|ею|єю|ою|є|еві|ем|єм|ів|їв|\'ю)$/u'; //http://uk.wikipedia.org/wiki/Іменник
private static $RVRE = '/^(.*?[аеиоуюяіїє])(.*)$/u';
private static $DERIVATIONAL = '/[^аеиоуюяіїє][аеиоуюяіїє]+[^аеиоуюяіїє]+[аеиоуюяіїє].*(?<=о)сть?$/u';
private static function s(&$s, $re, $to)
{
$orig = $s;
$s = preg_replace($re, $to, $s);
return $orig !== $s;
}
private static function m($s, $re)
{
return preg_match($re, $s);
}
public static function stem($word)
{
$word = mb_strtolower($word);
$stem = $word;
do {
if (!preg_match(self::$RVRE, $word, $p)) {
break;
}
$start = $p[1];
$RV = $p[2];
if (!$RV) {
break;
}
// Step 1
if (!self::s($RV, self::$PERFECTIVEGROUND, '')) {
self::s($RV, self::$REFLEXIVE, '');
if (self::s($RV, self::$ADJECTIVE, '')) {
self::s($RV, self::$PARTICIPLE, '');
} else {
if (!self::s($RV, self::$VERB, '')) {
self::s($RV, self::$NOUN, '');
}
}
}
// Step 2
self::s($RV, '/[и|i]$/u', '');
// Step 3
if (self::m($RV, self::$DERIVATIONAL)) {
self::s($RV, '/сть?$/u', '');
}
// Step 4
if (!self::s($RV, '/ь$/u', '')) {
self::s($RV, '/ейше?/u', '');
self::s($RV, '/нн$/u', 'н');
}
$stem = $start . $RV;
} while (FALSE);
return $stem;
}
}
@@ -0,0 +1 @@
["a", "ako", "ali", "bi", "bih", "bila", "bili", "bilo", "bio", "bismo", "biste", "biti", "bumo", "da", "do", " duž", "ga", "hoće", "hoćemo", "hoćete", "hoćeš", "hoću", "i", "iako", "ih", "ili", "iz", "ja", "je", "jedna", "jedne", "jedno", "jer", "jesam", "jesi", "jesmo", "jest", "jeste", "jesu", "jim", "joj", "još", "ju", "kada", "kako", "kao", "koja", "koje", "koji", "kojima", "koju", "kroz", "li", "me", "mene", "meni", "mi", "mimo", "moj", "moja", "moje", "mu", "na", "nad", "nakon", "nam", "nama", "nas", "naš", "naša", "naše", "našeg", "ne", "nego", "neka", "neki", "nekog", "neku", "nema", "netko", "neće", "nećemo", "nećete", "nećeš", "neću", "nešto", "ni", "nije", "nikoga", "nikoje", "nikoju", "nisam", "nisi", "nismo", "niste", "nisu", "njega", "njegov", "njegova", "njegovo", "njemu", "njezin", "njezina", "njezino", "njih", "njihov", "njihova", "njihovo", "njim", "njima", "njoj", "nju", "no", "o", "od", "odmah", "on", "ona", "oni", "ono", "ova", "pa", "pak", "po", "pod", "pored", "prije", "s", "sa", "sam", "samo", "se", "sebe", "sebi", "si", "smo", "ste", "su", "sve", "svi", "svog", "svoj", "svoja", "svoje", "svom", "ta", "tada", "taj", "tako", "te", "tebe", "tebi", "ti", "to", "toj", "tome", "tu", "tvoj", "tvoja", "tvoje", "u", "uz", "vam", "vama", "vas", "vaš", "vaša", "vaše", "već", "vi", "vrlo", "za", "zar", "će", "ćemo", "ćete", "ćeš", "ću", "što", "tijekom"]
@@ -0,0 +1 @@
["one", "also", "lets", "get", "still", "vs", "re", "our", "their", "couldn", "hadn't", "for", "these", "not", "themselves", "your", "won't", "which", "just", "o", "you're", "can", "shouldn't", "we", "at", "had", "and", "myself", "but", "you've", "having", "my", "was", "ve", "during", "it", "y", "she", "how", "haven't", "other", "aren't", "there", "doesn't", "he", "do", "you'll", "d", "where", "a", "hers", "are", "both", "i", "or", "itself", "while", "over", "have", "me", "him", "ain", "haven", "that", "down", "theirs", "shan", "what", "shan't", "them", "all", "mightn", "from", "when", "won", "then", "most", "wouldn", "now", "again", "why", "only", "by", "too", "don't", "herself", "wasn't", "with", "each", "above", "whom", "ll", "until", "her", "so", "who", "needn't", "ours", "after", "m", "isn't", "they", "weren't", "aren", "will", "doesn", "the", "any", "hasn't", "isn", "were", "his", "up", "yourself", "on", "out", "as", "off", "below", "own", "s", "into", "some", "t", "hasn", "between", "here", "should", "of", "in", "being", "mightn't", "mustn", "ourselves", "shouldn", "does", "an", "than", "mustn't", "yourselves", "to", "no", "about", "its", "more", "hadn", "himself", "further", "you", "is", "against", "once", "this", "should've", "nor", "did", "wasn", "she's", "weren", "has", "those", "been", "wouldn't", "don", "yours", "if", "few", "didn", "be", "needn", "couldn't", "that'll", "didn't", "same", "before", "ma", "because", "it's", "such", "very", "you'd", "doing", "through", "under", "am"]
@@ -0,0 +1 @@
["auront", "votre", "ils", "\u00e9tions", "et", "\u00e9tais", "avec", "elle", "nos", "\u00e9taient", "\u00e9tait", "soyez", "seront", "sommes", "eussions", "eus", "eurent", "aient", "ont", "ai", "tu", "aurais", "e\u00fbmes", "serais", "eu", "avait", "ce", "aie", "ayant", "avez", "aurez", "je", "serons", "sont", "aurons", "s", "\u00e9t\u00e9es", "soit", "\u00eates", "e\u00fbtes", "par", "qui", "y", "avaient", "ne", "vos", "auriez", "tes", "serai", "seraient", "\u00e9tiez", "te", "fus", "\u00e9tant", "fussions", "\u00e9t\u00e9s", "mon", "e\u00fbt", "d", "ayants", "avions", "f\u00fbmes", "eues", "eusses", "la", "n", "c", "lui", "est", "ayantes", "nous", "aies", "que", "aurions", "ces", "avons", "mes", "un", "le", "sa", "fusse", "aura", "leur", "eut", "eussent", "se", "les", "m", "ton", "\u00e9tantes", "serait", "ses", "t", "\u00e9t\u00e9", "une", "f\u00fbt", "fusses", "pas", "aux", "vous", "ayez", "ayons", "\u00e9tants", "es", "m\u00eame", "fut", "auraient", "eusse", "toi", "suis", "aviez", "aurai", "ayante", "seras", "ta", "sois", "f\u00fbtes", "auras", "qu", "\u00e9tante", "serions", "seriez", "pour", "ma", "on", "dans", "serez", "\u00e0", "son", "\u00e9t\u00e9e", "furent", "des", "l", "fussent", "ait", "notre", "sera", "me", "soyons", "il", "mais", "du", "en", "sur", "fussiez", "as", "ou", "avais", "de", "soient", "eue", "eux", "aurait", "eussiez", "au", "moi", "j"]
@@ -0,0 +1 @@
["anderer", "unseres", "keinem", "jener", "jenes", "keiner", "jedem", "anders", "da", "nichts", "sehr", "unseren", "den", "kein", "wie", "zu", "meine", "sondern", "ihm", "bei", "einige", "wollen", "denn", "ihres", "werde", "viel", "wenn", "eines", "uns", "welchem", "habe", "k\u00f6nnen", "mich", "und", "euren", "anderr", "dazu", "jedes", "kann", "an", "wir", "diesem", "was", "eure", "ihre", "wieder", "dann", "unser", "eurer", "in", "deine", "doch", "ist", "um", "demselben", "nach", "waren", "weil", "manchen", "dem", "ihn", "anderes", "ohne", "einen", "wollte", "jenem", "einiger", "seinen", "dessen", "jede", "mir", "keinen", "dasselbe", "k\u00f6nnte", "es", "hat", "oder", "\u00fcber", "deines", "ihr", "wird", "desselben", "vor", "meiner", "seines", "manchem", "hatten", "einigem", "anderen", "einmal", "diese", "meines", "ich", "also", "derselben", "hinter", "solchem", "war", "damit", "einem", "deiner", "aus", "seinem", "aller", "anderm", "sein", "nur", "einig", "dieselbe", "solchen", "weg", "haben", "hin", "deinen", "dass", "einigen", "da\u00df", "solche", "alle", "diesen", "im", "einiges", "du", "nicht", "zwischen", "w\u00fcrden", "das", "andere", "jenen", "sind", "die", "jetzt", "so", "dein", "vom", "bist", "dieser", "am", "dies", "des", "manche", "ihnen", "w\u00e4hrend", "allem", "indem", "aber", "musste", "dieselben", "eures", "gewesen", "ihrer", "welcher", "derselbe", "euer", "andern", "seiner", "dich", "denselben", "sie", "welchen", "dieses", "eurem", "unserem", "bis", "hier", "allen", "mancher", "wo", "einer", "auch", "gegen", "alles", "weiter", "nun", "keines", "keine", "meinen", "werden", "zwar", "der", "warst", "zur", "eine", "wirst", "ihren", "auf", "dir", "soll", "anderem", "als", "deinem", "durch", "von", "meinem", "jene", "ein", "mit", "unter", "zum", "bin", "hab", "derer", "jeden", "sollte", "w\u00fcrde", "welches", "ander", "er", "etwas", "sich", "manches", "welche", "seine", "jeder", "ins", "f\u00fcr", "solcher", "solches", "ihrem", "unsere", "will", "ob", "dort", "hatte", "mein", "sonst", "man", "muss", "noch", "machen", "selbst", "euch"]
@@ -0,0 +1 @@
["gli", "dove", "a", "fossero", "stiano", "alle", "avevano", "hanno", "mie", "sar\u00f2", "suoi", "stai", "questo", "un", "nei", "anche", "facessimo", "starebbe", "stemmo", "questa", "stesse", "sua", "dov", "o", "dallo", "ero", "dell", "starei", "stando", "negl", "fossi", "all", "sarai", "di", "suo", "far\u00f2", "tu", "si", "stavate", "facciano", "degli", "vostra", "avreste", "foste", "avranno", "ha", "facevo", "quelli", "sareste", "loro", "in", "degl", "come", "stanno", "ad", "lo", "avremo", "facciate", "avessi", "dalla", "vostro", "coi", "sugl", "con", "una", "quelle", "avuti", "eri", "eravamo", "eravate", "sono", "fanno", "stessero", "abbiamo", "chi", "sia", "alla", "nello", "tra", "nostra", "nostre", "avemmo", "sar\u00e0", "saremmo", "col", "al", "dei", "da", "facevano", "faceste", "mi", "facesse", "i", "avete", "\u00e8", "siate", "dai", "tuoi", "dal", "avevo", "farete", "avute", "allo", "avr\u00e0", "avuto", "farei", "io", "tua", "avevate", "negli", "l", "la", "faremo", "vostri", "saresti", "stette", "stavo", "avendo", "sarete", "stavamo", "fosse", "faranno", "perch\u00e9", "staremo", "voi", "delle", "noi", "stareste", "stava", "dagl", "se", "avrete", "quanto", "della", "nella", "sull", "sulle", "vi", "facesti", "li", "faceva", "facciamo", "miei", "sul", "fui", "avrai", "avessero", "avuta", "stiamo", "del", "stavi", "agl", "avevi", "erano", "uno", "abbiate", "stessi", "quanta", "staresti", "fosti", "sue", "stettero", "faremmo", "vostre", "nostri", "avevamo", "avrei", "abbia", "sulla", "le", "sarebbero", "quale", "quante", "quella", "ed", "nell", "tue", "far\u00e0", "fossimo", "farebbero", "siano", "aveste", "siamo", "saranno", "star\u00e0", "feci", "sugli", "lui", "fummo", "fai", "stetti", "ebbi", "ebbero", "furono", "ne", "non", "farai", "faccio", "pi\u00f9", "dagli", "avrebbe", "mio", "avesse", "era", "stia", "questi", "starai", "su", "il", "ho", "dalle", "nelle", "sui", "tutto", "ti", "star\u00f2", "fareste", "dello", "stesti", "facessero", "tuo", "aveva", "avessimo", "siete", "essendo", "staranno", "nostro", "ma", "c", "avresti", "stiate", "per", "queste", "stavano", "ci", "ebbe", "sto", "starete", "starebbero", "cui", "nel", "facevate", "fecero", "facendo", "e", "farebbe", "avr\u00f2", "quello", "avrebbero", "dall", "saremo", "ai", "avremmo", "fu", "fece", "stessimo", "contro", "sarebbe", "facevamo", "steste", "avesti", "faccia", "facessi", "agli", "quanti", "abbiano", "facevi", "sta", "facemmo", "faresti", "hai", "sei", "staremmo", "sullo", "mia", "sarei", "lei", "che", "tutti"]
@@ -0,0 +1 @@
["больше","может","много","более","ее","со","она","к","потому","и","хорошо","надо","не","же","по","есть","раз","конечно","у","нельзя","быть","кто","под","в","во","об","лучше","какой","даже","ему","до","я","почти","тем","вдруг","как","вы","них","да","но","вас","вам","сам","свою","там","нее","один","то","было","ну","эту","два","того","никогда","этот","чтобы","чего","нет","всего","меня","при","впрочем","этого","такой","после","нас","что","перед","ни","ведь","когда","им","ним","между","ж","а","из","наконец","вот","нибудь","куда","чуть","иногда","все","с","тогда","ты","тоже","ничего","себе","так","уже","они","тут","был","над","эти","какая","опять","этой","можно","совсем","него","ней","была","на","чем","для","еще","без","от","моя","потом","их","сейчас","этом","он","другой","про","здесь","три","были","будто","разве","только","всегда","уж","или","всех","мы","том","чтоб","если","где","за","тот","хоть","ей","зачем","через","о","себя","бы","мне","ли","всю","будет","мой","теперь","тебя","его"]
@@ -0,0 +1 @@
["hubierais", "sentido", "suya", "fu\u00e9semos", "estuvi\u00e9semos", "estar\u00e1s", "fuerais", "ha", "estar\u00e1n", "tuvi\u00e9ramos", "t\u00fa", "estuvi\u00e9ramos", "tuviesen", "habido", "hube", "os", "pero", "sentida", "habr\u00e9", "hayan", "otros", "sin", "suyos", "estuviste", "tanto", "tendr\u00eda", "tuvieron", "tuya", "lo", "hubieras", "que", "fueran", "estar\u00edamos", "sobre", "qu\u00e9", "se\u00e1is", "m\u00ed", "haya", "vosotras", "tuvierais", "\u00e9l", "tenidas", "ser\u00edas", "poco", "quien", "m\u00edas", "ti", "esto", "tiene", "hay\u00e1is", "otro", "estar\u00eda", "seremos", "suyas", "como", "ser\u00e9is", "me", "ni", "habr\u00e1", "tu", "algo", "una", "tenemos", "hab\u00edamos", "ten\u00edan", "estuvisteis", "sean", "hubieran", "la", "tuvieran", "tuvo", "soy", "era", "estadas", "estar\u00e1", "mucho", "tendr\u00e1", "estuviera", "fuiste", "fuese", "tendr\u00e9", "estos", "fu\u00e9ramos", "no", "ellas", "cual", "todo", "durante", "para", "est\u00e9", "los", "hemos", "habr\u00e9is", "contra", "habr\u00edas", "fuera", "ten\u00e9is", "estuvo", "con", "habr\u00eda", "cuando", "estad", "las", "estamos", "a", "tendr\u00e1s", "est\u00e1is", "nosotros", "estada", "esa", "tuvieseis", "hubisteis", "tened", "estaremos", "vuestras", "habr\u00edais", "fuesen", "te", "yo", "habr\u00edamos", "hubiesen", "habr\u00e1s", "y", "nosotras", "estuvieses", "tendr\u00e9is", "fueras", "m\u00edos", "vuestra", "estar\u00e9", "quienes", "tengas", "tuvi\u00e9semos", "entre", "mi", "hubiese", "desde", "tuviera", "ser\u00e1", "tendremos", "hubieron", "son", "estuvieseis", "estuvieras", "estando", "has", "tenido", "este", "teng\u00e1is", "muy", "un", "ten\u00eda", "est\u00e9is", "habr\u00edan", "tuve", "fui", "ten\u00edamos", "por", "tuvieras", "tuvieses", "estuvieran", "vuestro", "ser\u00e1n", "tambi\u00e9n", "porque", "nuestro", "ser\u00edan", "estuvimos", "fuimos", "estados", "se", "donde", "nuestra", "hubiera", "fueron", "somos", "est\u00e1n", "habidas", "sentidas", "m\u00edo", "todos", "esta", "fueses", "hayas", "tuvimos", "sois", "hab\u00edais", "algunos", "hubieses", "es", "hab\u00e9is", "de", "ella", "hab\u00edas", "teniendo", "del", "est\u00e1s", "ese", "est\u00e9n", "tus", "otra", "tuviese", "nada", "nuestras", "sus", "habr\u00e1n", "e", "hasta", "fue", "otras", "estuvierais", "est\u00e9s", "esas", "hubiste", "tienen", "ten\u00edas", "uno", "estemos", "tuyo", "tendr\u00edas", "su", "estuviese", "tendr\u00edais", "estaba", "eras", "fuisteis", "habiendo", "tuvisteis", "siente", "estuvieron", "vosotros", "tenidos", "estar\u00edais", "tengamos", "tendr\u00edamos", "hayamos", "hab\u00eda", "tenga", "estar\u00e9is", "estar", "mis", "hay", "tuyos", "sentidos", "ante", "estar\u00edan", "estuviesen", "tendr\u00edan", "habremos", "vuestros", "eso", "tengan", "estabas", "hubimos", "fueseis", "seamos", "hubieseis", "esos", "tenida", "sea", "el", "hubi\u00e9ramos", "en", "habidos", "he", "hubo", "hab\u00edan", "al", "estar\u00edas", "unos", "tuviste", "sintiendo", "ser\u00eda", "tendr\u00e1n", "ten\u00edais", "algunas", "estuve", "s\u00ed", "nuestros", "o", "est\u00e1bamos", "eres", "habida", "nos", "hubi\u00e9semos", "antes", "estaban", "eran", "m\u00e1s", "han", "\u00e9ramos", "estabais", "tuyas", "seas", "les", "ser\u00e1s", "tengo", "ellos", "ser\u00e9", "sentid", "ser\u00edamos", "estas", "muchos", "erais", "estoy", "suyo", "est\u00e1", "tienes", "le", "ser\u00edais", "estado", "m\u00eda", "ya"]
File diff suppressed because one or more lines are too long
@@ -0,0 +1,16 @@
<?php
namespace TeamTNT\TNTSearch\Support;
abstract class AbstractTokenizer
{
static protected $pattern = '';
public function getPattern()
{
if (empty(static::$pattern)) {
throw new \LogicException("Tokenizer must define split \$pattern value");
} else {
return static::$pattern;
}
}
}
@@ -0,0 +1,11 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class BigramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
public function tokenize($text, $stopwords = [])
{
$ngramTokenizer = new NGramTokenizer(2, 2);
return $ngramTokenizer->tokenize($text, $stopwords);
}
}
@@ -0,0 +1,164 @@
<?php
namespace TeamTNT\TNTSearch\Support;
use ArrayIterator;
use Countable;
use IteratorAggregate;
use Traversable;
class Collection implements Countable, IteratorAggregate
{
protected $items = [];
public function __construct($items = [])
{
$this->items = $items;
}
public function forget($key)
{
unset($this->items[$key]);
}
/**
* @param callable $callback
*
* @return Collection
*/
public function each(callable $callback)
{
foreach ($this->items as $key => $item) {
if ($callback($item, $key) === false) {
break;
}
}
return $this;
}
/**
* @param callable|null $callback
*
* @return static
*/
public function filter(callable $callback = null)
{
if ($callback) {
$return = [];
foreach ($this->items as $key => $value) {
if ($callback($value, $key)) {
$return[$key] = $value;
}
}
return new static($return);
}
return new static(array_filter($this->items));
}
/**
* @return bool
*/
public function isEmpty()
{
return empty($this->items);
}
/**
* @param callable $callback
*
* @return static
*/
public function map(callable $callback)
{
$keys = array_keys($this->items);
$items = array_map($callback, $this->items, $keys);
return new static(array_combine($keys, $items));
}
/**
* @param callable $callback
* @param null $initial
*
* @return mixed
*/
public function reduce(callable $callback, $initial = null)
{
return array_reduce($this->items, $callback, $initial);
}
public function get($key)
{
return $this->items[$key];
}
/**
* @param $value
* @param null $key
*
* @return array
*/
public function pluck($value, $key = null)
{
return array_column($this->items, $value, $key);
}
/**
* @param $glue
*
* @return string
*/
public function implode($glue)
{
return implode($glue, $this->items);
}
/**
* @return int
*/
public function count(): int
{
return count($this->items);
}
/**
* @param int $offset
* @param int $length
*
* @return static
*/
public function slice($offset, $length = null)
{
return new static(array_slice($this->items, $offset, $length, true));
}
/**
* @param int $limit
* @return static
*/
public function take($limit)
{
return $this->slice(0, abs($limit));
}
/**
* @return ArrayIterator
*/
public function getIterator(): Traversable
{
return new ArrayIterator($this->items);
}
/**
* @return array
*/
public function toArray()
{
return $this->items;
}
}
@@ -0,0 +1,23 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class EdgeNgramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
protected static $pattern = '/[\s,\.]+/';
public function tokenize($text, $stopwords = [])
{
$text = mb_strtolower($text);
$ngrams = [];
$splits = preg_split($this->getPattern(), $text, -1, PREG_SPLIT_NO_EMPTY);
foreach ($splits as $split) {
for ($i = 2; $i <= strlen($split); $i++) {
$ngrams[] = mb_substr($split, 0, $i);
}
}
return $ngrams;
}
}
@@ -0,0 +1,116 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class Expression
{
protected $operatorQueue;
protected $tokenQueue;
public function toPostfix($exp)
{
$postfix = [];
$stack = [];
$tokens = $this->lex($exp);
foreach ($tokens as $token) {
if ($this->isOperand($token)) {
$postfix[] = $token;
} else {
if ($token == ")") {
while (($top = array_pop($stack)) != "(" && !empty($top)) {
$postfix[] = $top;
}
} else {
while (
count($stack) && !(end($stack) == "(") &&
($this->priority(end($stack)) >= $this->priority($token))
) {
$postfix[] = array_pop($stack);
}
$stack[] = $token;
}
}
}
while (!empty($stack)) {
$postfix[] = array_pop($stack);
}
return $postfix;
}
public function isOperand($str)
{
if (
($str == "|") || ($str == "&") || ($str == "~") ||
($str == "(") || ($str == ")")
) {
return false;
}
return true;
}
public function isOperator($str)
{
return !$this->isOperand($str);
}
public function priority($operator)
{
$priority = 0;
if ($operator == ("&")) {
$priority = 2;
}
if ($operator == "~") {
$priority = 3;
}
if ($operator == "|") {
$priority = 1;
}
if ($operator == "(" || $operator == ")") {
$priority = 4;
}
return $priority;
}
public function lex($string)
{
$bad = [' or ', ' -', ' '];
$good = ['|', '~', '&'];
$string = str_replace($bad, $good, $string);
$string = mb_strtolower($string);
$tokens = [];
$token = "";
foreach (str_split($string) as $char) {
if ($this->isOperator($char)) {
if ($token) {
$tokens[] = $token;
}
$tokens[] = $char;
$token = "";
} else {
$token .= $char;
}
}
if ($token) {
$tokens[] = $token;
}
return $tokens;
}
}
@@ -0,0 +1,12 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class FivegramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
public function tokenize($text, $stopwords = [])
{
$ngramTokenizer = new NGramTokenizer(5, 5);
return $ngramTokenizer->tokenize($text, $stopwords);
}
}
@@ -0,0 +1,12 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class FourgramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
public function tokenize($text, $stopwords = [])
{
$ngramTokenizer = new NGramTokenizer(4, 4);
return $ngramTokenizer->tokenize($text, $stopwords);
}
}
@@ -0,0 +1,220 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class Highlighter
{
protected $options = [
'simple' => false,
'wholeWord' => true,
'caseSensitive' => false,
'stripLinks' => false,
'tagOptions' => [
// 'class' => 'search-term', // Example
// 'title' => 'You searched for this.', // Example
// 'data-toggle' => 'tooltip', // Example
]
];
protected $tokenizer;
public function __construct(TokenizerInterface $tokenizer = null)
{
if (!empty($tokenizer)) {
$this->tokenizer = $tokenizer;
} else {
$this->tokenizer = new Tokenizer;
}
}
/**
* @param $text
* @param $needle
* @param string $tag
* @param array $options
*
* @return string
*/
public function highlight($text, $needle, $tag = 'em', $options = [])
{
$this->options = array_merge($this->options, $options);
$tagAttributes = '';
if (count($this->options['tagOptions'])) {
foreach ($this->options['tagOptions'] as $attr => $value) {
$tagAttributes .= $attr . '="' . $value . '" ';
}
$tagAttributes = ' ' . trim($tagAttributes);
}
$highlight = '<' . $tag . $tagAttributes .'>\1</' . $tag . '>';
$needle = preg_split($this->tokenizer->getPattern(), $needle, -1, PREG_SPLIT_NO_EMPTY);
// Select pattern to use
if ($this->options['simple']) {
$pattern = '#(%s)#';
$sl_pattern = '#(%s)#';
} else {
$pattern = '#(?!<.*?)(%s)(?![^<>]*?>)#';
$sl_pattern = '#<a\s(?:.*?)>(%s)</a>#';
}
// Add Forgotten Unicode
$pattern .= 'u';
// Case sensitivity
if (!($this->options['caseSensitive'])) {
$pattern .= 'i';
$sl_pattern .= 'i';
}
$needle = (array) $needle;
foreach ($needle as $needle_s) {
$needle_s = preg_quote($needle_s);
// Escape needle with optional whole word check
if ($this->options['wholeWord']) {
$needle_s = '\b' . $needle_s . '\b';
}
// Strip links
if ($this->options['stripLinks']) {
$sl_regex = sprintf($sl_pattern, $needle_s);
$text = preg_replace($sl_regex, '\1', $text);
}
$regex = sprintf($pattern, $needle_s);
$text = preg_replace($regex, $highlight, $text);
}
return $text;
}
/**
* find the locations of each of the words
* Nothing exciting here. The array_unique is required
* unless you decide to make the words unique before passing in
*
* @param $words
* @param $fulltext
*
* @return array
*/
public function _extractLocations($words, $fulltext)
{
$locations = array();
foreach ($words as $word) {
$wordlen = mb_strlen($word);
$loc = mb_stripos($fulltext, $word);
while ($loc !== false) {
$locations[] = $loc;
$loc = mb_stripos($fulltext, $word, $loc + $wordlen);
}
}
$locations = array_unique($locations);
sort($locations);
return $locations;
}
/**
* Work out which is the most relevant portion to display
* This is done by looping over each match and finding the smallest distance between two found
* strings. The idea being that the closer the terms are the better match the snippet would be.
* When checking for matches we only change the location if there is a better match.
* The only exception is where we have only two matches in which case we just take the
* first as will be equally distant.
*
* @param $locations
* @param $prevcount
*
* @return int
*/
public function _determineSnipLocation($locations, $prevcount)
{
if (!isset($locations[0])) {
return -1;
}
// If we only have 1 match we dont actually do the for loop so set to the first
$startpos = $locations[0];
$loccount = count($locations);
$smallestdiff = PHP_INT_MAX;
// If we only have 2 skip as its probably equally relevant
if (count($locations) > 2) {
// skip the first as we check 1 behind
for ($i = 1; $i < $loccount; $i++) {
if ($i == $loccount - 1) {
// at the end
$diff = $locations[$i] - $locations[$i - 1];
} else {
$diff = $locations[$i + 1] - $locations[$i];
}
if ($smallestdiff > $diff) {
$smallestdiff = $diff;
$startpos = $locations[$i];
}
}
}
$startpos = $startpos > $prevcount ? $startpos - $prevcount : 0;
return $startpos;
}
/**
* 1/6 ratio on prevcount tends to work pretty well and puts the terms
* in the middle of the extract
*
* @param $words
* @param $fulltext
* @param int $rellength
* @param int $prevcount
* @param string $indicator
*
* @return bool|string
*/
public function extractRelevant($words, $fulltext, $rellength = 300, $prevcount = 50, $indicator = '...')
{
$words = preg_split($this->tokenizer->getPattern(), $words, -1, PREG_SPLIT_NO_EMPTY);
$textlength = mb_strlen($fulltext);
if ($textlength <= $rellength) {
return $fulltext;
}
$locations = $this->_extractLocations($words, $fulltext);
$startpos = $this->_determineSnipLocation($locations, $prevcount);
// if we are going to snip too much...
if ($textlength - $startpos < $rellength) {
$startpos = $startpos - ($textlength - $startpos) / 2;
}
// in case no match is found, reset position for proper math below
if ($startpos == -1) {
$startpos = 0;
}
$reltext = mb_substr($fulltext, $startpos, $rellength);
preg_match_all($this->tokenizer->getPattern(), $reltext, $offset, PREG_OFFSET_CAPTURE);
// since PREG_OFFSET_CAPTURE returns offset in bytes we have to use mb_strlen(substr()) hack here
$last = mb_strlen(substr($reltext, 0, end($offset[0])[1]));
$first = mb_strlen(substr($reltext, 0, $offset[0][0][1]));
// if no match is found, just return first $rellength characters without the last word
if (empty($locations)) {
return mb_substr($reltext, 0, $last) . $indicator;
}
// check to ensure we dont snip the last word if thats the match
if ($startpos + $rellength < $textlength) {
$reltext = mb_substr($reltext, 0, $last) . $indicator; // remove last word
}
// If we trimmed from the front add ...
if ($startpos != 0) {
$reltext = $indicator . mb_substr($reltext, $first + 1); // remove first word
}
return $reltext;
}
}
@@ -0,0 +1,34 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class NGramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
public $min_gram = 3;
public $max_gram = 3;
public function __construct($min_gram = 3, $max_gram = 3)
{
$this->min_gram = $min_gram;
$this->max_gram = $max_gram;
}
protected static $pattern = '/[\s,\.]+/';
public function tokenize($text, $stopwords = [])
{
$text = mb_strtolower($text);
$ngrams = [];
$splits = preg_split($this->getPattern(), $text, -1, PREG_SPLIT_NO_EMPTY);
foreach ($splits as $split) {
for ($currentGram = $this->min_gram; $currentGram <= $this->max_gram; $currentGram++) {
for ($i = 0; $i <= strlen($split) - $currentGram; $i++) {
$ngrams[] = mb_substr($split, $i, $currentGram);
}
}
}
return $ngrams;
}
}
@@ -0,0 +1,14 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class ProductTokenizer extends AbstractTokenizer implements TokenizerInterface
{
static protected $pattern = '/[\s,\.]+/';
public function tokenize($text, $stopwords = [])
{
$text = mb_strtolower($text);
$split = preg_split($this->getPattern(), $text, -1, PREG_SPLIT_NO_EMPTY);
return array_diff($split, $stopwords);
}
}
@@ -0,0 +1,14 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class Tokenizer extends AbstractTokenizer implements TokenizerInterface
{
static protected $pattern = '/[^\p{L}\p{N}\p{Pc}\p{Pd}@]+/u';
public function tokenize($text, $stopwords = [])
{
$text = mb_strtolower($text);
$split = preg_split($this->getPattern(), $text, -1, PREG_SPLIT_NO_EMPTY);
return array_diff($split, $stopwords);
}
}
@@ -0,0 +1,9 @@
<?php
namespace TeamTNT\TNTSearch\Support;
interface TokenizerInterface
{
public function tokenize($text, $stopwords);
public function getPattern();
}
@@ -0,0 +1,12 @@
<?php
namespace TeamTNT\TNTSearch\Support;
class TrigramTokenizer extends AbstractTokenizer implements TokenizerInterface
{
public function tokenize($text, $stopwords = [])
{
$ngramTokenizer = new NGramTokenizer(3, 3);
return $ngramTokenizer->tokenize($text, $stopwords);
}
}
@@ -0,0 +1,155 @@
<?php
namespace TeamTNT\TNTSearch;
class TNTFuzzyMatch
{
public function norm($vec)
{
$norm = 0;
$components = count($vec);
for ($i = 0; $i < $components; $i++) {
$norm += $vec[$i] * $vec[$i];
}
return sqrt($norm);
}
public function dot($vec1, $vec2)
{
$prod = 0;
$components = count($vec1);
for ($i = 0; $i < $components; $i++) {
$prod += ($vec1[$i] * $vec2[$i]);
}
return $prod;
}
public function wordToVector($word)
{
$alphabet = "aAbBcCčČćĆdDđĐeEfFgGhHiIjJkKlLmMnNoOpPqQrRsSšŠtTvVuUwWxXyYzZžŽ1234567890'+ /";
$result = [];
foreach (str_split($word) as $w) {
$result[] = strpos($alphabet, $w) + 1000000;
}
return $result;
}
public function angleBetweenVectors($a, $b)
{
$denominator = ($this->norm($a) * $this->norm($b));
if ($denominator == 0) {
return 0;
}
return $this->dot($a, $b) / $denominator;
}
public function hasCommonSubsequence($pattern, $str)
{
$pattern = mb_strtolower($pattern);
$str = mb_strtolower($str);
$j = 0;
$patternLength = strlen($pattern);
$strLength = strlen($str);
for ($i = 0; $i < $strLength && $j < $patternLength; $i++) {
if ($pattern[$j] == $str[$i]) {
$j++;
}
}
return ($j == $patternLength);
}
public function makeVectorSameLength($str, $pattern)
{
$j = 0;
$max = max(count($pattern), count($str));
$a = [];
$b = [];
for ($i = 0; $i < $max && $j < $max; $i++) {
if (isset($pattern[$j]) && isset($str[$i]) && $pattern[$j] == $str[$i]) {
$j++;
$b[] = $str[$i];
} else {
$b[] = 0;
}
}
return $b;
}
public function fuzzyMatchFromFile($pattern, $path)
{
$res = [];
$lines = fopen($path, "r");
if ($lines) {
while (!feof($lines)) {
$line = rtrim(fgets($lines, 4096));
if ($this->hasCommonSubsequence($pattern, $line)) {
$res[] = $line;
}
}
fclose($lines);
}
$paternVector = $this->wordToVector($pattern);
$sorted = [];
foreach ($res as $caseSensitiveWord) {
$word = mb_strtolower(trim($caseSensitiveWord));
$wordVector = $this->wordToVector($word);
$normalizedPaternVector = $this->makeVectorSameLength($wordVector, $paternVector);
$angle = $this->angleBetweenVectors($wordVector, $normalizedPaternVector);
if (strpos($word, $pattern) !== false) {
$angle += 0.2;
}
$sorted[$caseSensitiveWord] = $angle;
}
arsort($sorted);
return $sorted;
}
public function fuzzyMatch($pattern, $items)
{
$res = [];
foreach ($items as $item) {
if ($this->hasCommonSubsequence($pattern, $item)) {
$res[] = $item;
}
}
$paternVector = $this->wordToVector($pattern);
$sorted = [];
foreach ($res as $word) {
$word = trim($word);
$wordVector = $this->wordToVector($word);
$normalizedPaternVector = $this->makeVectorSameLength($wordVector, $paternVector);
$angle = $this->angleBetweenVectors($wordVector, $normalizedPaternVector);
if (strpos($word, $pattern) !== false) {
$angle += 0.2;
}
$sorted[$word] = $angle;
}
arsort($sorted);
return $sorted;
}
}
@@ -0,0 +1,91 @@
<?php
namespace TeamTNT\TNTSearch;
use PDO;
use TeamTNT\TNTSearch\Indexer\TNTGeoIndexer;
use TeamTNT\TNTSearch\Support\Collection;
class TNTGeoSearch extends TNTSearch
{
protected $earthRadius = 6371;
/**
* Distance is in KM
*/
public function findNearest($currentLocation, $distance, $limit = 10)
{
$startTimer = microtime(true);
$res = $this->buildQuery($currentLocation, $distance, $limit);
$stopTimer = microtime(true);
return [
'ids' => $res->pluck('doc_id'),
'distances' => $res->pluck('distance'),
'hits' => $res->count(),
'execution_time' => round($stopTimer - $startTimer, 7) * 1000 ." ms"
];
}
public function buildQuery($currentLocation, $distance, $limit)
{
$query = "
SELECT doc_id, longitude, latitude,
:CUR_sin_lat * sin_lat + :CUR_cos_lat * cos_lat * (cos_lng * :CUR_cos_lng + sin_lng * :CUR_sin_lng) AS distance
FROM locations AS l
JOIN (
SELECT :latpoint AS latpoint, :longpoint AS longpoint,
:radius AS radius, 111.045 AS distance_unit
) AS p
WHERE l.latitude
BETWEEN p.latpoint - (p.radius / p.distance_unit)
AND p.latpoint + (p.radius / p.distance_unit)
AND l.longitude
BETWEEN p.longpoint - (p.radius / (p.distance_unit * :CUR_cos_lat))
AND p.longpoint + (p.radius / (p.distance_unit * :CUR_cos_lat))
ORDER BY distance DESC
LIMIT :limit";
$stmtDoc = $this->index->prepare($query);
$cur_lat = $currentLocation['latitude'];
$cur_lng = $currentLocation['longitude'];
$CUR_cos_lat = cos($cur_lat * pi() / 180);
$CUR_sin_lat = sin($cur_lat * pi() / 180);
$CUR_cos_lng = cos($cur_lng * pi() / 180);
$CUR_sin_lng = sin($cur_lng * pi() / 180);
$stmtDoc->bindValue(':latpoint', $cur_lat);
$stmtDoc->bindValue(':longpoint', $cur_lng);
$stmtDoc->bindValue(':radius', $distance);
$stmtDoc->bindValue(':CUR_cos_lat', $CUR_cos_lat);
$stmtDoc->bindValue(':CUR_sin_lat', $CUR_sin_lat);
$stmtDoc->bindValue(':CUR_cos_lng', $CUR_cos_lng);
$stmtDoc->bindValue(':CUR_sin_lng', $CUR_sin_lng);
$stmtDoc->bindValue(':limit', $limit);
$stmtDoc->execute();
$locations = new Collection($stmtDoc->fetchAll(PDO::FETCH_ASSOC));
$locations = $locations->map(function ($location) use ($distance) {
$location['distance'] = acos($location['distance']) * $this->earthRadius;
if ($location['distance'] <= $distance) {
return $location;
}
});
return $locations;
}
public function getIndex()
{
$indexer = new TNTGeoIndexer;
$indexer->inMemory = false;
$indexer->setIndex($this->index);
return $indexer;
}
}
@@ -0,0 +1,513 @@
<?php
namespace TeamTNT\TNTSearch;
use PDO;
use TeamTNT\TNTSearch\Exceptions\IndexNotFoundException;
use TeamTNT\TNTSearch\Indexer\TNTIndexer;
use TeamTNT\TNTSearch\Stemmer\NoStemmer;
use TeamTNT\TNTSearch\Support\Collection;
use TeamTNT\TNTSearch\Support\Expression;
use TeamTNT\TNTSearch\Support\Highlighter;
use TeamTNT\TNTSearch\Support\Tokenizer;
use TeamTNT\TNTSearch\Support\TokenizerInterface;
class TNTSearch
{
public $config;
public $asYouType = false;
public $maxDocs = 500;
public $tokenizer = null;
public $index = null;
public $stemmer = null;
public $fuzziness = false;
public $fuzzy_prefix_length = 2;
public $fuzzy_max_expansions = 50;
public $fuzzy_distance = 2;
protected $dbh = null;
/**
* @param array $config
*
* @see https://github.com/teamtnt/tntsearch#examples
*/
public function loadConfig(array $config)
{
$this->config = $config;
$this->config['storage'] = rtrim($this->config['storage'], '/').'/';
}
public function __construct()
{
$this->tokenizer = new Tokenizer;
}
/**
* @param PDO $dbh
*/
public function setDatabaseHandle(PDO $dbh)
{
$this->dbh = $dbh;
}
/**
* @param string $indexName
* @param boolean $disableOutput
*
* @return TNTIndexer
*/
public function createIndex($indexName, $disableOutput = false)
{
$indexer = new TNTIndexer;
$indexer->loadConfig($this->config);
$indexer->disableOutput = $disableOutput;
if ($this->dbh) {
$indexer->setDatabaseHandle($this->dbh);
}
return $indexer->createIndex($indexName);
}
/**
* @param string $indexName
*
* @throws IndexNotFoundException
*/
public function selectIndex($indexName)
{
$pathToIndex = $this->config['storage'].$indexName;
if (!file_exists($pathToIndex)) {
throw new IndexNotFoundException("Index {$pathToIndex} does not exist", 1);
}
$this->index = new PDO('sqlite:'.$pathToIndex);
$this->index->setAttribute(PDO::ATTR_ERRMODE, PDO::ERRMODE_EXCEPTION);
$this->setStemmer();
$this->setTokenizer();
}
/**
* @param string $phrase
* @param int $numOfResults
*
* @return array
*/
public function search($phrase, $numOfResults = 100)
{
$startTimer = microtime(true);
$keywords = $this->breakIntoTokens($phrase);
$keywords = new Collection($keywords);
$keywords = $keywords->map(function ($keyword) {
return $this->stemmer->stem($keyword);
});
$tfWeight = 1;
$dlWeight = 0.5;
$docScores = [];
$count = $this->totalDocumentsInCollection();
foreach ($keywords as $index => $term) {
$isLastKeyword = ($keywords->count() - 1) == $index;
$df = $this->totalMatchingDocuments($term, $isLastKeyword);
$idf = log($count / max(1, $df));
foreach ($this->getAllDocumentsForKeyword($term, false, $isLastKeyword) as $document) {
$docID = $document['doc_id'];
$tf = $document['hit_count'];
$num = ($tfWeight + 1) * $tf;
$denom = $tfWeight
* ((1 - $dlWeight) + $dlWeight)
+ $tf;
$score = $idf * ($num / $denom);
$docScores[$docID] = isset($docScores[$docID]) ?
$docScores[$docID] + $score : $score;
}
}
arsort($docScores);
$docs = new Collection($docScores);
$totalHits = $docs->count();
$docs = $docs->map(function ($doc, $key) {
return $key;
})->take($numOfResults);
$stopTimer = microtime(true);
if ($this->isFileSystemIndex()) {
return $this->filesystemMapIdsToPaths($docs)->toArray();
}
return [
'ids' => array_keys($docs->toArray()),
'hits' => $totalHits,
'execution_time' => round($stopTimer - $startTimer, 7) * 1000 ." ms"
];
}
/**
* @param string $phrase
* @param int $numOfResults
*
* @return array
*/
public function searchBoolean($phrase, $numOfResults = 100)
{
$stack = [];
$startTimer = microtime(true);
$expression = new Expression;
$postfix = $expression->toPostfix("|".$phrase);
foreach ($postfix as $token) {
if ($token == '&') {
$left = array_pop($stack);
$right = array_pop($stack);
if (is_string($left)) {
$left = $this->getAllDocumentsForKeyword($this->stemmer->stem($left), true)
->pluck('doc_id');
}
if (is_string($right)) {
$right = $this->getAllDocumentsForKeyword($this->stemmer->stem($right), true)
->pluck('doc_id');
}
if (is_null($left)) {
$left = [];
}
if (is_null($right)) {
$right = [];
}
$stack[] = array_values(array_intersect($left, $right));
} else
if ($token == '|') {
$left = array_pop($stack);
$right = array_pop($stack);
if (is_string($left)) {
$left = $this->getAllDocumentsForKeyword($this->stemmer->stem($left), true)
->pluck('doc_id');
}
if (is_string($right)) {
$right = $this->getAllDocumentsForKeyword($this->stemmer->stem($right), true)
->pluck('doc_id');
}
if (is_null($left)) {
$left = [];
}
if (is_null($right)) {
$right = [];
}
$stack[] = array_unique(array_merge($left, $right));
} else
if ($token == '~') {
$left = array_pop($stack);
if (is_string($left)) {
$left = $this->getAllDocumentsForWhereKeywordNot($this->stemmer->stem($left), true)
->pluck('doc_id');
}
if (is_null($left)) {
$left = [];
}
$stack[] = $left;
} else {
$stack[] = $token;
}
}
if (count($stack)) {
$docs = new Collection($stack[0]);
} else {
$docs = new Collection;
}
$docs = $docs->take($numOfResults);
$stopTimer = microtime(true);
if ($this->isFileSystemIndex()) {
return $this->filesystemMapIdsToPaths($docs)->toArray();
}
return [
'ids' => $docs->toArray(),
'hits' => $docs->count(),
'execution_time' => round($stopTimer - $startTimer, 7) * 1000 ." ms"
];
}
/**
* @param $keyword
* @param bool $noLimit
* @param bool $isLastKeyword
*
* @return Collection
*/
public function getAllDocumentsForKeyword($keyword, $noLimit = false, $isLastKeyword = false)
{
$word = $this->getWordlistByKeyword($keyword, $isLastKeyword);
if (!isset($word[0])) {
return new Collection([]);
}
if ($this->fuzziness) {
return $this->getAllDocumentsForFuzzyKeyword($word, $noLimit);
}
return $this->getAllDocumentsForStrictKeyword($word, $noLimit);
}
/**
* @param $keyword
* @param bool $noLimit
*
* @return Collection
*/
public function getAllDocumentsForWhereKeywordNot($keyword, $noLimit = false)
{
$word = $this->getWordlistByKeyword($keyword);
if (!isset($word[0])) {
return new Collection([]);
}
$query = "SELECT * FROM doclist WHERE doc_id NOT IN (SELECT doc_id FROM doclist WHERE term_id = :id) GROUP BY doc_id ORDER BY hit_count DESC LIMIT {$this->maxDocs}";
if ($noLimit) {
$query = "SELECT * FROM doclist WHERE doc_id NOT IN (SELECT doc_id FROM doclist WHERE term_id = :id) GROUP BY doc_id ORDER BY hit_count DESC";
}
$stmtDoc = $this->index->prepare($query);
$stmtDoc->bindValue(':id', $word[0]['id']);
$stmtDoc->execute();
return new Collection($stmtDoc->fetchAll(PDO::FETCH_ASSOC));
}
/**
* @param $keyword
* @param bool $isLastWord
*
* @return int
*/
public function totalMatchingDocuments($keyword, $isLastWord = false)
{
$occurance = $this->getWordlistByKeyword($keyword, $isLastWord);
if (isset($occurance[0])) {
return $occurance[0]['num_docs'];
}
return 0;
}
/**
* @param $keyword
* @param bool $isLastWord
*
* @return array
*/
public function getWordlistByKeyword($keyword, $isLastWord = false)
{
$searchWordlist = "SELECT * FROM wordlist WHERE term like :keyword LIMIT 1";
$stmtWord = $this->index->prepare($searchWordlist);
if ($this->asYouType && $isLastWord) {
$searchWordlist = "SELECT * FROM wordlist WHERE term like :keyword ORDER BY length(term) ASC, num_hits DESC LIMIT 1";
$stmtWord = $this->index->prepare($searchWordlist);
$stmtWord->bindValue(':keyword', mb_strtolower($keyword)."%");
} else {
$stmtWord->bindValue(':keyword', mb_strtolower($keyword));
}
$stmtWord->execute();
$res = $stmtWord->fetchAll(PDO::FETCH_ASSOC);
if ($this->fuzziness && !isset($res[0])) {
return $this->fuzzySearch($keyword);
}
return $res;
}
/**
* @param $keyword
*
* @return array
*/
public function fuzzySearch($keyword)
{
$prefix = mb_substr($keyword, 0, $this->fuzzy_prefix_length);
$searchWordlist = "SELECT * FROM wordlist WHERE term like :keyword ORDER BY num_hits DESC LIMIT {$this->fuzzy_max_expansions}";
$stmtWord = $this->index->prepare($searchWordlist);
$stmtWord->bindValue(':keyword', mb_strtolower($prefix)."%");
$stmtWord->execute();
$matches = $stmtWord->fetchAll(PDO::FETCH_ASSOC);
$resultSet = [];
foreach ($matches as $match) {
$distance = levenshtein($match['term'], $keyword);
if ($distance <= $this->fuzzy_distance) {
$match['distance'] = $distance;
$resultSet[] = $match;
}
}
// Sort the data by distance, and than by num_hits
$distance = [];
$hits = [];
foreach ($resultSet as $key => $row) {
$distance[$key] = $row['distance'];
$hits[$key] = $row['num_hits'];
}
array_multisort($distance, SORT_ASC, $hits, SORT_DESC, $resultSet);
return $resultSet;
}
public function totalDocumentsInCollection()
{
return $this->getValueFromInfoTable('total_documents');
}
public function getStemmer()
{
return $this->stemmer;
}
public function setStemmer()
{
$stemmer = $this->getValueFromInfoTable('stemmer');
if ($stemmer) {
$this->stemmer = new $stemmer;
} else {
$this->stemmer = isset($this->config['stemmer']) ? new $this->config['stemmer'] : new NoStemmer;
}
}
public function setTokenizer()
{
$tokenizer = $this->getValueFromInfoTable('tokenizer');
if ($tokenizer) {
$this->tokenizer = new $tokenizer;
} else {
$this->tokenizer = isset($this->config['tokenizer']) ? new $this->config['tokenizer'] : new Tokenizer;
}
}
/**
* @return bool
*/
public function isFileSystemIndex()
{
return $this->getValueFromInfoTable('driver') == 'filesystem';
}
public function getValueFromInfoTable($value)
{
$query = "SELECT * FROM info WHERE key = '$value'";
$docs = $this->index->query($query);
if ($ret = $docs->fetch(PDO::FETCH_ASSOC)) {
return $ret['value'];
}
return null;
}
public function filesystemMapIdsToPaths($docs)
{
$query = "SELECT * FROM filemap WHERE id in (".$docs->implode(', ').");";
$res = $this->index->query($query)->fetchAll(PDO::FETCH_ASSOC);
return $docs->map(function ($key) use ($res) {
$index = array_search($key, array_column($res, 'id'));
return $res[$index];
});
}
public function info($str)
{
echo $str."\n";
}
public function breakIntoTokens($text)
{
return $this->tokenizer->tokenize($text);
}
/**
* @param $text
* @param $needle
* @param string $tag
* @param array $options
*
* @return string
*/
public function highlight($text, $needle, $tag = 'em', $options = [])
{
$hl = new Highlighter($this->tokenizer);
return $hl->highlight($text, $needle, $tag, $options);
}
public function snippet($words, $fulltext, $rellength = 300, $prevcount = 50, $indicator = '...')
{
$hl = new Highlighter($this->tokenizer);
return $hl->extractRelevant($words, $fulltext, $rellength, $prevcount, $indicator);
}
/**
* @return TNTIndexer
*/
public function getIndex()
{
$indexer = new TNTIndexer;
$indexer->inMemory = false;
$indexer->setIndex($this->index);
$indexer->setStemmer($this->stemmer);
$indexer->setTokenizer($this->tokenizer);
return $indexer;
}
/**
* @param $words
* @param $noLimit
*
* @return Collection
*/
private function getAllDocumentsForFuzzyKeyword($words, $noLimit)
{
$binding_params = implode(',', array_fill(0, count($words), '?'));
$query = "SELECT * FROM doclist WHERE term_id in ($binding_params) ORDER BY CASE term_id";
$order_counter = 1;
foreach ($words as $word) {
$query .= " WHEN ".$word['id']." THEN ".$order_counter++;
}
$query .= " END";
if (!$noLimit) {
$query .= " LIMIT {$this->maxDocs}";
}
$stmtDoc = $this->index->prepare($query);
$ids = null;
foreach ($words as $word) {
$ids[] = $word['id'];
}
$stmtDoc->execute($ids);
return new Collection($stmtDoc->fetchAll(PDO::FETCH_ASSOC));
}
/**
* @param $word
* @param $noLimit
*
* @return Collection
*/
private function getAllDocumentsForStrictKeyword($word, $noLimit)
{
$query = "SELECT * FROM doclist WHERE term_id = :id ORDER BY hit_count DESC LIMIT {$this->maxDocs}";
if ($noLimit) {
$query = "SELECT * FROM doclist WHERE term_id = :id ORDER BY hit_count DESC";
}
$stmtDoc = $this->index->prepare($query);
$stmtDoc->bindValue(':id', $word[0]['id']);
$stmtDoc->execute();
return new Collection($stmtDoc->fetchAll(PDO::FETCH_ASSOC));
}
}
@@ -0,0 +1,115 @@
<?php
use TeamTNT\TNTSearch\TNTFuzzyMatch;
class TNTFuzzyMatchTest extends PHPUnit\Framework\TestCase
{
public function __construct()
{
$this->fm = new TNTFuzzyMatch;
parent::__construct();
}
public function testNorm()
{
$vector = [3, 4];
$normalized = $this->fm->norm($vector);
$this->assertEquals(5, $normalized);
$vector = [1, 2, 3, 4, 5];
$normalized = $this->fm->norm($vector);
$this->assertEquals(7.416198487095663, $normalized);
}
public function testDot()
{
$vector1 = [1, 2, -5];
$vector2 = [4, 8, 1];
$product = $this->fm->dot($vector1, $vector2);
$this->assertEquals(15, $product);
}
public function testWordToVector()
{
$word = "TNT";
$vector = $this->fm->wordToVector($word);
$this->assertEquals($vector, [1000055, 1000039, 1000055]);
}
public function testAngleBetweenVectors()
{
$vector1 = [1, 2, 3];
$vector2 = [4, 5, 6];
$angle = $this->fm->angleBetweenVectors($vector1, $vector2);
$this->assertEquals(0.97463184619707621, $angle);
}
public function testHasCommonSubsequence()
{
$pattern1 = "tnsarh";
$pattern2 = "ntnsearch";
$res1 = $this->fm->hasCommonSubsequence($pattern1, 'tntsearch');
$res2 = $this->fm->hasCommonSubsequence($pattern2, 'tntsearch');
$this->assertEquals($res1, true);
$this->assertEquals($res2, false);
}
public function testMakeVectorSameLength()
{
$wordVector = $this->fm->wordToVector("tntsearch");
$patternVector = $this->fm->wordToVector("tnth");
$res = $this->fm->makeVectorSameLength($wordVector, $patternVector);
$this->assertEquals([1000054, 1000038, 1000054, 0, 0, 0, 0, 0, 1000026], $res);
}
public function testFuzzyMatchFromFile()
{
$res = $this->fm->fuzzyMatchFromFile('search', __DIR__.'/_files/english_wordlist_2k.txt');
$equal = bccomp($res['search'], 1.2, 2);
$this->assertEquals(0, $equal);
$equal = bccomp($res['research'], 1.06, 2);
$this->assertEquals(0, $equal);
}
public function testFuzzyMatchFromFileFunction()
{
$res = fuzzyMatchFromFile('search', __DIR__.'/_files/english_wordlist_2k.txt');
$equal = bccomp($res['search'], 1.2, 2);
$this->assertEquals(0, $equal);
$equal = bccomp($res['research'], 1.06, 2);
$this->assertEquals(0, $equal);
}
public function testFuzzyMatch()
{
$res = $this->fm->fuzzyMatch('search', ['search', 'research', 'something']);
$equal = bccomp($res['search'], 1.2, 2);
$this->assertEquals(0, $equal);
$equal = bccomp($res['research'], 1.06, 2);
$this->assertEquals(0, $equal);
}
public function testFuzzyMatchFunction()
{
$res = fuzzyMatch('search', ['search', 'research', 'something']);
$equal = bccomp($res['search'], 1.2, 2);
$this->assertEquals(0, $equal);
$equal = bccomp($res['research'], 1.06, 2);
$this->assertEquals(0, $equal);
}
}
@@ -0,0 +1,41 @@
<?php
use TeamTNT\TNTSearch\TNTGeoSearch;
class TNTGeoSearchTest extends PHPUnit\Framework\TestCase
{
protected $indexName = "cities-geo.index";
protected $config = [
'storage' => __DIR__.'/_files/'
];
/**
* If we're located in Munich, lets find 2 nearest cities around 50km
*/
public function testFindNearest()
{
$currentLocation = [
'longitude' => 11.576124,
'latitude' => 48.137154
];
$distance = 50; //km
$citiesIndex = new TNTGeoSearch();
$citiesIndex->loadConfig($this->config);
$citiesIndex->selectIndex($this->indexName);
$cities = $citiesIndex->findNearest($currentLocation, $distance, 2);
$this->assertEquals([9389, 9407], $cities['ids']);
$this->assertEquals(2, $cities['hits']);
}
public function tearDown(): void
{
if (file_exists(__DIR__.'/../_files/'.$this->indexName)) {
unlink(__DIR__.'/../_files/'.$this->indexName);
}
}
}
@@ -0,0 +1,348 @@
<?php
use TeamTNT\TNTSearch\Exceptions\IndexNotFoundException;
use TeamTNT\TNTSearch\TNTSearch;
class TNTSearchTest extends PHPUnit\Framework\TestCase
{
protected $indexName = "testIndex";
protected $config = [
'driver' => 'sqlite',
'database' => __DIR__.'/_files/articles.sqlite',
'host' => 'localhost',
'username' => 'testUser',
'password' => 'testPass',
'storage' => __DIR__.'/_files/',
'stemmer' => \TeamTNT\TNTSearch\Stemmer\PorterStemmer::class
];
public function testLoadConfig()
{
$tnt = new TNTSearch();
$tnt->loadConfig($this->config);
$this->assertArrayHasKey('driver', $tnt->config);
$this->assertArrayHasKey('database', $tnt->config);
$this->assertArrayHasKey('host', $tnt->config);
$this->assertArrayHasKey('username', $tnt->config);
$this->assertArrayHasKey('password', $tnt->config);
$this->assertArrayHasKey('storage', $tnt->config);
$this->assertArrayHasKey('stemmer', $tnt->config);
}
public function testCreateIndex()
{
$tnt = new TNTSearch();
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$this->assertInstanceOf('TeamTNT\TNTSearch\Indexer\TNTIndexer', $indexer);
$this->assertFileExists($indexer->getStoragePath().$this->indexName);
}
public function testSearchBoolean()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->searchBoolean('romeo juliet queen');
$this->assertEquals([7], $res['ids']);
$res = $tnt->searchBoolean('Hamlet or Macbeth');
$this->assertEquals([3, 4, 1, 2], $res['ids']);
$this->assertEquals(4, $res['hits']);
$res = $tnt->searchBoolean('juliet ~well');
$this->assertEquals([5, 6, 7, 8, 10], $res['ids']);
$res = $tnt->searchBoolean('juliet ~romeo');
$this->assertEquals([10], $res['ids']);
$res = $tnt->searchBoolean('hamlet ~king');
$this->assertEquals([2], $res['ids']);
$res = $tnt->searchBoolean('hamlet superman');
$this->assertEquals([], $res['ids']);
$res = $tnt->searchBoolean('hamlet or superman');
$this->assertEquals([1, 2], $res['ids']);
$res = $tnt->searchBoolean('hamlet');
$this->assertEquals([1, 2], $res['ids']);
$res = $tnt->searchBoolean('eldred ~bar');
$this->assertEquals([11], $res['ids']);
$res = $tnt->searchBoolean('Eldred ~bar');
$this->assertEquals([11], $res['ids']);
}
/**
* https://github.com/teamtnt/tntsearch/issues/60
*/
public function testTotalDocumentCountOnIndexUpdate()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$this->assertEquals(12, $tnt->totalDocumentsInCollection());
$index = $tnt->getIndex();
//first we test if the total number of documents will decrease
$index->delete(12);
$this->assertEquals(11, $tnt->totalDocumentsInCollection());
//now we try with a document that does not exist, the total number should increase for 1
$index->update(1234, ['id' => '1234', 'title' => 'updated title', 'article' => 'updated article']);
$this->assertEquals(12, $tnt->totalDocumentsInCollection());
}
public function testPrimaryKeyIncludedInResult()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->includePrimaryKey();
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->search(3);
$this->assertEquals([3], $res['ids']);
}
public function testPrimaryKeyNotIncludedInResult()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->search(3);
$this->assertEquals([], $res['ids']);
}
public function testIndexUpdate()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$index = $tnt->getIndex();
$count = $index->countWordInWordList('titl');
$this->assertTrue($count == 0, 'Word titl should be 0');
$index->insert(['id' => '11', 'title' => 'new title', 'article' => 'new article']);
$count = $index->countWordInWordList('titl');
$this->assertEquals(1, $count, 'Word titl should be 1');
$docCount = $index->countDocHitsInWordList('juliet');
$this->assertEquals(6, $docCount, 'Juliet should occur in 6 documents');
$index->insert(['id' => '12', 'title' => 'juliet', 'article' => 'new article about juliet']);
$count = $index->countWordInWordList('juliet');
$this->assertEquals(9, $count, 'Word juliet should be 9');
$docCount = $index->countDocHitsInWordList('juliet');
$this->assertEquals(7, $docCount, 'Juliet should occur in 7 documents');
$index->delete(12);
$count = $index->countWordInWordList('juliet');
$this->assertEquals(7, $count, 'Word juliet should be 7 after delete');
$docCount = $index->countDocHitsInWordList('juliet');
$this->assertEquals(6, $docCount, 'Juliet should occur in 6 documents after delete');
$count = $index->countWordInWordList('romeo');
$this->assertEquals(5, $count, 'Word romeo should be 5');
$index->update(11, ['id' => '11', 'title' => 'romeo', 'article' => 'new article about romeo']);
$count = $index->countWordInWordList('romeo');
$this->assertEquals(7, $count, 'Word romeo should be 7');
}
public function testMultipleSearch()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->search('Othello');
$this->assertEmpty($res['ids']);
$this->assertEquals(12, $tnt->totalDocumentsInCollection());
$index = $tnt->getIndex();
$count = $index->countWordInWordList('Othello');
$this->assertTrue($count == 0, 'Word Othello should be 0');
$index->insert(['id' => '13', 'title' => 'Othello', 'article' => 'For she had eyes and chose me.']);
$count = $index->countWordInWordList('Othello');
$this->assertEquals(1, $count, 'Word Othello should be 1');
$this->assertEquals(13, $tnt->totalDocumentsInCollection());
$res = $tnt->search('Othello');
$this->assertEquals([13], $res['ids']);
}
public function testAsYouType()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$tnt->asYouType = true;
$res = $tnt->search('k');
$this->assertEquals([1], $res['ids']);
}
public function testHits()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->search('juliet');
$this->assertEquals(6, $res['hits']);
}
public function testFuzzySearch()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$tnt->fuzziness = true;
$res = $tnt->search('juleit');
$this->assertEquals("9", $res['ids'][0]);
$res = $tnt->search('quen');
$this->assertEquals("7", $res['ids'][0]);
$res = $tnt->search('asdf');
$this->assertEquals([], $res['ids']);
}
public function testFuzzySearchMultipleWordsFound()
{
$tnt = new TNTSearch();
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$index = $tnt->getIndex();
$index->insert(['id' => '14', 'title' => '199x', 'article' => 'Nineties with the x...']);
$index->insert(['id' => '15', 'title' => '199y', 'article' => 'Nineties with the y...']);
$tnt->fuzziness = true;
$res = $tnt->search('199');
$this->assertContains(14, $res['ids']);
$this->assertContains(15, $res['ids']);
}
public function testIndexDoesNotExistException()
{
$this->expectException(IndexNotFoundException::class);
$this->expectExceptionCode(1);
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$tnt->selectIndex('IndexThatDoesNotExist');
}
public function testStemmerIsSetOnNewIndexesBasedOnConfig()
{
$config = $this->config;
$config['stemmer'] = \TeamTNT\TNTSearch\Stemmer\GermanStemmer::class;
$tnt = new TNTSearch();
$tnt->loadConfig($config);
$tnt->createIndex($this->indexName);
$tnt->selectIndex($this->indexName);
$this->assertInstanceOf(\TeamTNT\TNTSearch\Stemmer\GermanStemmer::class, $tnt->getStemmer());
}
public function testDefaultStemmerIsSetOnNewIndexesIfNoneConfigured()
{
$config = $this->config;
unset($config['stemmer']);
$tnt = new TNTSearch();
$tnt->loadConfig($config);
$tnt->createIndex($this->indexName);
$tnt->selectIndex($this->indexName);
$this->assertInstanceOf(\TeamTNT\TNTSearch\Stemmer\NoStemmer::class, $tnt->getStemmer());
}
public function tearDown(): void
{
if (file_exists(__DIR__."/".$this->indexName)) {
unlink(__DIR__."/".$this->indexName);
}
}
}
@@ -0,0 +1 @@
first document
@@ -0,0 +1 @@
second document
@@ -0,0 +1 @@
third document
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,2 @@
*
!.gitignore
File diff suppressed because one or more lines are too long
@@ -0,0 +1,59 @@
<?php
use TeamTNT\TNTSearch\Classifier\TNTClassifier;
class TNTClassifierTest extends PHPUnit\Framework\TestCase
{
public function testPredictSpamHam()
{
$smsJSON = file_get_contents(__DIR__.'/../_files/sms-texts.json');
$sms = json_decode($smsJSON);
$training = 0.80;
$classifier = new TNTClassifier();
for ($i = 0; $i <= count($sms) * $training; $i++) {
$classifier->learn($sms[$i]->message, $sms[$i]->label);
}
$guessCount = 0;
$counter = 0;
for ($i = round(count($sms) * $training); $i < count($sms); $i++) {
$counter++;
$guess = $classifier->predict($sms[$i]->message);
if ($guess['label'] == $sms[$i]->label) {
$guessCount++;
}
}
$precision = number_format(($guessCount * 100 / $counter), 4);
$this->assertGreaterThanOrEqual(98, $precision);
}
public function testPredictClass()
{
$classifier = new TNTClassifier();
$classifier->learn("chinese beijing chinese", "c");
$classifier->learn("chinese chinese shangai", "c");
$classifier->learn("chinese macao", "c");
$classifier->learn("tokyo japan chinese", "j");
$guess = $classifier->predict("chinese chinese chinese tokyo japan");
$this->assertEquals("c", $guess['label']);
}
public function testPredictClass2()
{
$classifier = new TNTClassifier();
$classifier->learn("A great game", "Sports");
$classifier->learn("The election was over", "Not sports");
$classifier->learn("Very clean match", "Sports");
$classifier->learn("A clean but forgettable game", "Sports");
$guess = $classifier->predict("It was a close election");
$this->assertEquals("Not sports", $guess['label']);
}
}
@@ -0,0 +1,32 @@
<?php
use TeamTNT\TNTSearch\Indexer\TNTGeoIndexer;
use TeamTNT\TNTSearch\Indexer\TNTIndexer;
class TNTGeoIndexerTest extends PHPUnit\Framework\TestCase
{
protected $indexName = "cities-geo.index";
protected $config = [
'driver' => 'sqlite',
'database' => __DIR__.'/../_files/cities.sqlite',
'host' => 'localhost',
'username' => 'testUser',
'password' => 'testPass',
'storage' => __DIR__.'/../_files/'
];
public function testGeoIndexCreation()
{
$geoIndex = new TNTGeoIndexer;
$geoIndex->disableOutput = true;
$geoIndex->loadConfig($this->config);
$geoIndex->createIndex($this->indexName);
$geoIndex->query('SELECT id, longitude, latitude FROM cities;');
$geoIndex->run();
$indexPath = __DIR__.'/../_files/'.$this->indexName;
$this->assertTrue(file_exists($indexPath));
}
}
@@ -0,0 +1,190 @@
<?php
use TeamTNT\TNTSearch\Indexer\TNTIndexer;
use TeamTNT\TNTSearch\Support\AbstractTokenizer;
use TeamTNT\TNTSearch\Support\TokenizerInterface;
use TeamTNT\TNTSearch\TNTSearch;
class TNTIndexerTest extends PHPUnit\Framework\TestCase
{
protected $indexName = "testIndex";
protected $config = [
'driver' => 'sqlite',
'database' => __DIR__.'/../_files/articles.sqlite',
'host' => 'localhost',
'username' => 'testUser',
'password' => 'testPass',
'storage' => __DIR__.'/../_files/',
'tokenizer' => TeamTNT\TNTSearch\Support\ProductTokenizer::class
];
public function testSearch()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$tnt->asYouType = true;
$res = $tnt->search('Juliet');
//the most relevant doc has the id 9
$this->assertEquals("9", $res['ids'][0]);
$res = $tnt->search('Queen Mab');
$this->assertEquals([7], $res['ids']);
}
public function testIndexFromFileSystem()
{
$config = [
'driver' => 'filesystem',
'storage' => __DIR__.'/../_files/',
'location' => __DIR__.'/../_files/articles/',
'extension' => 'txt'
];
$tnt = new TNTSearch;
$tnt->loadConfig($config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->disableOutput = true;
$indexer->run();
$tnt->selectIndex($this->indexName);
$index = $tnt->getIndex();
$count = $index->countWordInWordList('document');
$this->assertTrue($count == 3, 'Word document should be 3');
$this->assertEquals('TeamTNT\TNTSearch\Stemmer\NoStemmer', get_class($tnt->getStemmer()));
}
public function testIfCroatianStemmerIsSet()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->setLanguage('croatian');
$indexer->disableOutput = true;
$indexer->run();
$this->index = new PDO('sqlite:'.$this->config['storage'].$this->indexName);
$query = "SELECT * FROM info WHERE key = 'stemmer'";
$docs = $this->index->query($query);
$value = $docs->fetch(PDO::FETCH_ASSOC)['value'];
$this->assertEquals('TeamTNT\TNTSearch\Stemmer\CroatianStemmer', $value);
$tnt->selectIndex($this->indexName);
$this->assertEquals('TeamTNT\TNTSearch\Stemmer\CroatianStemmer', get_class($tnt->getStemmer()));
}
public function testIfGermanStemmerIsSet()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->setLanguage('german');
$indexer->disableOutput = true;
$indexer->run();
$this->index = new PDO('sqlite:'.$this->config['storage'].$this->indexName);
$query = "SELECT * FROM info WHERE key = 'stemmer'";
$docs = $this->index->query($query);
$value = $docs->fetch(PDO::FETCH_ASSOC)['value'];
$this->assertEquals('TeamTNT\TNTSearch\Stemmer\GermanStemmer', $value);
$tnt->selectIndex($this->indexName);
$this->assertEquals('TeamTNT\TNTSearch\Stemmer\GermanStemmer', get_class($tnt->getStemmer()));
}
public function testBuildTrigrams()
{
$indexer = new TNTIndexer;
$trigrams = $indexer->buildTrigrams('created');
$this->assertEquals('__c _cr cre rea eat ate ted ed_ d__', $trigrams);
$trigrams = $indexer->buildTrigrams('mood');
$this->assertEquals('__m _mo moo ood od_ d__', $trigrams);
$trigrams = $indexer->buildTrigrams('death');
$this->assertEquals('__d _de dea eat ath th_ h__', $trigrams);
$trigrams = $indexer->buildTrigrams('behind');
$this->assertEquals('__b _be beh ehi hin ind nd_ d__', $trigrams);
$trigrams = $indexer->buildTrigrams('usually');
$this->assertEquals('__u _us usu sua ual all lly ly_ y__', $trigrams);
$trigrams = $indexer->buildTrigrams('created');
$this->assertEquals('__c _cr cre rea eat ate ted ed_ d__', $trigrams);
}
public function tearDown(): void
{
if (file_exists(__DIR__.'/../_files/'.$this->indexName)) {
unlink(__DIR__.'/../_files/'.$this->indexName);
}
}
public function testSetTokenizer()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->query('SELECT id, title, article FROM articles;');
$indexer->setTokenizer(new SomeTokenizer);
$indexer->disableOutput = true;
$indexer->run();
$this->assertInstanceOf(TokenizerInterface::class, $indexer->tokenizer);
$res = $indexer->breakIntoTokens('Canon 70-200');
$this->assertContains("canon", $res);
$this->assertContains("70-200", $res);
}
public function testCustomPrimaryKey()
{
$tnt = new TNTSearch;
$tnt->loadConfig($this->config);
$indexer = $tnt->createIndex($this->indexName);
$indexer->setPrimaryKey('post_id');
$indexer->disableOutput = true;
$indexer->query('SELECT * FROM posts;');
$indexer->run();
$tnt->selectIndex($this->indexName);
$res = $tnt->search('second');
//the most relevant doc has the id 9
$this->assertEquals("2", $res['ids'][0]);
}
}
class SomeTokenizer extends AbstractTokenizer implements TokenizerInterface
{
static protected $pattern = '/[\s,\.]+/';
public function tokenize($text, $stopwords = [])
{
return preg_split($this->getPattern(), mb_strtolower($text), -1, PREG_SPLIT_NO_EMPTY);
}
}
@@ -0,0 +1,107 @@
<?php
use TeamTNT\TNTSearch\KeywordExtraction\Rake;
class RakeTest extends PHPUnit\Framework\TestCase
{
public function __construct()
{
$this->rake = new Rake;
parent::__construct();
}
public function testExtractKeywords()
{
$text = "A scoop of ice cream";
$actual = $this->rake->extractKeywords($text);
$expected = ["ice cream" => 4, "scoop" => 1];
$this->assertEquals($expected, $actual);
}
public function testExtractKeywords2()
{
$text = "Compatibility of systems of linear constraints over the set of natural
numbers. Criteria of compatibility of a system of linear Diophantine
equations, strict inequations, and nonstrict inequations are considered.
Upper bounds for components of a minimal set of solutions and algorithms
of construction of minimal generating sets of solutions for all types of
systems are given. These criteria and the corresponding algorithms for
constructing a minimal supporting set of solutions can be used in solving
all the considered types of systems and systems of mixed types.";
$actual = $this->rake->extractKeywords($text);
$this->assertEquals(8.666666666666666, $actual["minimal generating sets"], '', 0.0001);
$this->assertEquals(8.5, $actual["linear diophantine equations"], '', 0.0001);
$this->assertEquals(7.666666666666666, $actual["minimal supporting set"], '', 0.0001);
}
public function testTokenize()
{
$expected = ["a", "scoop", "of", "ice", "cream"];
$actual = $this->rake->tokenize('A scoop of ice cream');
$this->assertEquals($expected, $actual);
}
public function testGenerateCandidateKeywords()
{
$text = "A scoop of ice cream";
$expected = [["scoop"], ["ice", "cream"]];
$actual = $this->rake->generateCandidateKeywords($text);
$this->assertEquals($expected, $actual);
}
public function testWordDegree()
{
$text = "A scoop of ice cream";
$phraseList = $this->rake->generateCandidateKeywords($text);
$degree = $this->rake->wordDegree("scoop", $phraseList);
$this->assertEquals($degree, 1);
$degree = $this->rake->wordDegree("ice", $phraseList);
$this->assertEquals($degree, 2);
$degree = $this->rake->wordDegree("cream", $phraseList);
$this->assertEquals($degree, 2);
}
public function testWordDegree2()
{
$text = "Compatibility of systems of linear constraints over the set of natural numbers of Criteria of compatibility of a system of linear Diophantine equations, strict inequations, and nonstrict inequations are considered. Upper bounds for components of a minimal set of solutions and algorithms of construction of minimal generating sets of solutions for all types of systems are given. These criteria and the corresponding algorithms for constructing a minimal supporting set of solutions can be used in solving all the considered types of systems and systems of mixed types.";
$phraseList = $this->rake->generateCandidateKeywords($text);
$degree = $this->rake->wordDegree("set", $phraseList);
$this->assertEquals(6, $degree);
$degree = $this->rake->wordDegree("natural", $phraseList);
$this->assertEquals(2, $degree);
}
public function testWordFrequency()
{
$text = "Compatibility of systems of linear constraints over the set of natural numbers of Criteria of compatibility of a system of linear Diophantine equations, strict inequations, and nonstrict inequations are considered. Upper bounds for components of a minimal set of solutions and algorithms of construction of minimal generating sets of solutions for all types of systems are given. These criteria and the corresponding algorithms for constructing a minimal supporting set of solutions can be used in solving all the considered types of systems and systems of mixed types.";
$phraseList = $this->rake->generateCandidateKeywords($text);
$frequency = $this->rake->wordFrequency("systems", $phraseList);
$this->assertEquals(4, $frequency);
}
public function testCalculateWordScores()
{
$text = "A scoop of ice cream";
$expected = ["scoop" => 1, "ice" => 2, "cream" => 2];
$phraseList = $this->rake->generateCandidateKeywords($text);
$actual = $this->rake->calculateWordScores($phraseList);
$this->assertEquals($expected, $actual);
}
}
@@ -0,0 +1,50 @@
<?php
use TeamTNT\TNTSearch\Spell\JaroWinklerDistance;
class JaroWinklerDistanceTest extends PHPUnit\Framework\TestCase
{
public function __construct()
{
$this->sd = new JaroWinklerDistance;
parent::__construct();
}
public function testJaro()
{
$d = $this->sd->jaro('DWAYNE', 'DUANE');
$this->assertEqualsWithDelta(0.822, $d, 0.001);
$d = $this->sd->jaro("MARTHA", "MARHTA");
$this->assertEqualsWithDelta(0.944444, $d, 0.001);
$d = $this->sd->jaro("DIXON", "DICKSONX");
$this->assertEqualsWithDelta(0.766667, $d, 0.001);
$d = $this->sd->jaro("JELLYFISH", "SMELLYFISH");
$this->assertEqualsWithDelta(0.896296, $d, 0.001);
}
public function testGetDistance()
{
$d = $this->sd->getDistance("al", "al");
$this->assertEquals(1.0, $d);
$d = $this->sd->getDistance("martha", "marhta");
$this->assertGreaterThan(0.961, $d);
$this->assertLessThan(0.962, $d);
$d = $this->sd->getDistance("jones", "johnson");
$this->assertTrue($d > 0.832 && $d < 0.833);
$d = $this->sd->getDistance("dwayne", "duane");
$this->assertTrue($d > 0.84 && $d < 0.841);
$d = $this->sd->getDistance("dixon", "dicksonx");
$this->assertTrue($d > 0.813 && $d < 0.814);
$d = $this->sd->getDistance("fvie", "ten");
$this->assertTrue($d == 0);
$d1 = $this->sd->getDistance("zac ephron", "zac efron");
$d2 = $this->sd->getDistance("zac ephron", "kai ephron");
$this->assertTrue($d1 > $d2);
$d1 = $this->sd->getDistance("brittney spears", "britney spears");
$d2 = $this->sd->getDistance("brittney spears", "brittney startzman");
$this->assertTrue($d1 > $d2);
}
}
@@ -0,0 +1,59 @@
<?php
use TeamTNT\TNTSearch\Stemmer\CroatianStemmer;
class CroatianStemmerTest extends PHPUnit\Framework\TestCase
{
public function testIstakniSlogotvornoR()
{
$stemmer = new CroatianStemmer;
$this->assertEquals("cRveno", $stemmer->istakniSlogotvornoR("crveno"));
$this->assertEquals("tvRdo", $stemmer->istakniSlogotvornoR("tvrdo"));
$this->assertEquals("vRt", $stemmer->istakniSlogotvornoR("vrt"));
}
public function testImaSamoglasnik()
{
$stemmer = new CroatianStemmer;
$this->assertTrue($stemmer->imaSamoglasnik("test"));
$this->assertTrue($stemmer->imaSamoglasnik("vrt"));
$this->assertFalse($stemmer->imaSamoglasnik("dgk"));
}
public function testTransformiraj()
{
$stemmer = new CroatianStemmer;
$this->assertEquals("ginekologa", $stemmer->transformiraj("ginekolozi"));
$this->assertEquals("ujak", $stemmer->transformiraj("ujaci"));
$this->assertEquals("policajca", $stemmer->transformiraj("policajaca"));
}
public function testKorjenuj()
{
$stemmer = new CroatianStemmer;
$this->assertEquals("njem", $stemmer->korjenuj("njemu"));
$this->assertEquals("stisk", $stemmer->korjenuj("stiska"));
$this->assertEquals("jasn", $stemmer->korjenuj("jasno"));
$this->assertEquals("kalibr", $stemmer->korjenuj("kalibra"));
$this->assertEquals("zagrijavanj", $stemmer->korjenuj("zagrijavanje"));
$this->assertEquals("biznis", $stemmer->korjenuj("biznisom"));
$this->assertEquals("razgovara", $stemmer->korjenuj("razgovarati"));
$this->assertEquals("najbogat", $stemmer->korjenuj("najbogatijih"));
}
public function testStem()
{
$stemmer = new CroatianStemmer;
$this->assertEquals("biti", $stemmer->stem("biti"));
$this->assertEquals("njem", $stemmer->stem("njemu"));
$this->assertEquals("stisk", $stemmer->stem("stiska"));
$this->assertEquals("jasn", $stemmer->stem("jasno"));
$this->assertEquals("kalibr", $stemmer->stem("kalibra"));
$this->assertEquals("zagrijavanj", $stemmer->stem("zagrijavanje"));
$this->assertEquals("biznis", $stemmer->stem("biznisom"));
$this->assertEquals("razgovara", $stemmer->stem("razgovarati"));
$this->assertEquals("najbogat", $stemmer->stem("najbogatijih"));
$this->assertEquals("čćžšđ", $stemmer->stem("čćžšđ"));
}
}
@@ -0,0 +1,13 @@
<?php
use TeamTNT\TNTSearch\Stemmer\FrenchStemmer;
class FrenchStemmerTest extends PHPUnit\Framework\TestCase
{
public function testStem()
{
$this->assertSame('abaiss', FrenchStemmer::stem('abaissant'));
$this->assertSame('abandon', FrenchStemmer::stem('abandonnés'));
$this->assertSame(FrenchStemmer::stem('frontières'), FrenchStemmer::stem('frontière'));
}
}
@@ -0,0 +1,14 @@
<?php
use TeamTNT\TNTSearch\Stemmer\GermanStemmer;
class GermanStemmerTest extends PHPUnit\Framework\TestCase
{
public function testStem()
{
$stemmer = new GermanStemmer;
$this->assertEquals("vergnug", $stemmer->stem("vergnüglich"));
$this->assertEquals("unfallversicherungstrag", $stemmer->stem("Unfallversicherungsträger"));
}
}
@@ -0,0 +1,21 @@
<?php
use TeamTNT\TNTSearch\Stemmer\PolishStemmer;
class PolishStemmerTest extends PHPUnit\Framework\TestCase
{
public function testStem()
{
$stemmer = new PolishStemmer;
$this->assertEquals("czujnik", $stemmer->stem("czujnikami"));
$this->assertEquals("kabel", $stemmer->stem("kabelek"));
$this->assertEquals("mocniej", $stemmer->stem("najmocniejszy"));
$this->assertEquals("przekaźnik", $stemmer->stem("przekaźnikowy"));
$this->assertEquals("instaluj", $stemmer->stem("instalujesz"));
$this->assertEquals("instaluj", $stemmer->stem("instalujesz"));
$this->assertEquals("ciekaw", $stemmer->stem("ciekawie"));
$this->assertEquals("przekaźnik", $stemmer->stem("przekaźników"));
$this->assertEquals("moduł", $stemmer->stem("modułu"));
$this->assertEquals($stemmer->stem("modułów"), $stemmer->stem("moduły"));
}
}
@@ -0,0 +1,30 @@
<?php
use TeamTNT\TNTSearch\Stemmer\PorterStemmer;
class PorterStemmerTest extends PHPUnit\Framework\TestCase
{
public function testStem()
{
$stemmer = new PorterStemmer;
$this->assertEquals("test", $stemmer->stem("testing"));
$this->assertEquals("sourc", $stemmer->stem("source"));
$this->assertEquals("code", $stemmer->stem("code"));
$this->assertEquals("is", $stemmer->stem("is"));
$this->assertEquals("funni", $stemmer->stem("funny"));
}
public function testAgainstDictionary()
{
$vocabulary = explode("\n", file_get_contents(__DIR__."/porter/input.txt"));
$expected = explode("\n", file_get_contents(__DIR__."/porter/output.txt"));
$stemmer = new PorterStemmer;
foreach ($vocabulary as $key => $word) {
$stem = $stemmer->stem(trim($word));
$this->assertEquals(trim($expected[$key]), $stem);
}
}
}

Some files were not shown because too many files have changed in this diff Show More