usr.sbin/makemandb/custom_apropos_tokenizer.c

1.6      tnn /*	$NetBSD: custom_apropos_tokenizer.c,v 1.6 2023/08/07 20:35:21 tnn Exp $	*/
1.1  abhinav /*
1.1  abhinav ** 2006 September 30
1.1  abhinav **
1.1  abhinav ** The author disclaims copyright to this source code.  In place of
1.1  abhinav ** a legal notice, here is a blessing:
1.1  abhinav **
1.1  abhinav **    May you do good and not evil.
1.1  abhinav **    May you find forgiveness for yourself and forgive others.
1.1  abhinav **    May you share freely, never taking more than you give.
1.1  abhinav **
1.1  abhinav *************************************************************************
1.1  abhinav ** Implementation of the full-text-search tokenizer that implements
1.1  abhinav ** a Porter stemmer.
1.1  abhinav */
1.1  abhinav
1.1  abhinav /*
1.1  abhinav ** The code in this file is only compiled if:
1.1  abhinav **
1.1  abhinav **     * The FTS3 module is being built as an extension
1.1  abhinav **       (in which case SQLITE_CORE is not defined), or
1.1  abhinav **
1.1  abhinav **     * The FTS3 module is being built into the core of
1.1  abhinav **       SQLite (in which case SQLITE_ENABLE_FTS3 is defined).
1.1  abhinav */
1.1  abhinav
1.1  abhinav #include <assert.h>
1.1  abhinav #include <ctype.h>
1.1  abhinav #include <stdlib.h>
1.1  abhinav #include <stdio.h>
1.1  abhinav #include <string.h>
1.1  abhinav
1.1  abhinav #include "custom_apropos_tokenizer.h"
1.1  abhinav #include "fts3_tokenizer.h"
1.1  abhinav #include "nostem.c"
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Class derived from sqlite3_tokenizer
1.1  abhinav  */
1.1  abhinav typedef struct custom_apropos_tokenizer {
1.1  abhinav 	sqlite3_tokenizer base;	/* Base class */
1.1  abhinav } custom_apropos_tokenizer;
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Class derived from sqlite3_tokenizer_cursor
1.1  abhinav  */
1.1  abhinav typedef struct custom_apropos_tokenizer_cursor {
1.1  abhinav 	sqlite3_tokenizer_cursor base;
1.1  abhinav 	const char *zInput;	/* input we are tokenizing */
1.1  abhinav 	size_t nInput;		/* size of the input */
1.1  abhinav 	size_t iOffset;		/* current position in zInput */
1.1  abhinav 	size_t iToken;		/* index of next token to be returned */
1.1  abhinav 	char *zToken;		/* storage for current token */
1.1  abhinav 	size_t nAllocated;		/* space allocated to zToken buffer */
1.1  abhinav } custom_apropos_tokenizer_cursor;
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Create a new tokenizer instance.
1.1  abhinav  */
1.1  abhinav static int
1.1  abhinav aproposPorterCreate(int argc, const char *const * argv,
1.1  abhinav     sqlite3_tokenizer ** ppTokenizer)
1.1  abhinav {
1.1  abhinav 	custom_apropos_tokenizer *t;
1.1  abhinav 	t = calloc(1, sizeof(*t));
1.1  abhinav 	if (t == NULL)
1.1  abhinav 		return SQLITE_NOMEM;
1.1  abhinav 	*ppTokenizer = &t->base;
1.1  abhinav 	return SQLITE_OK;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Destroy a tokenizer
1.1  abhinav  */
1.5      rin static int
1.1  abhinav aproposPorterDestroy(sqlite3_tokenizer * pTokenizer)
1.1  abhinav {
1.1  abhinav 	free(pTokenizer);
1.1  abhinav 	return SQLITE_OK;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Prepare to begin tokenizing a particular string.  The input
1.1  abhinav  * string to be tokenized is zInput[0..nInput-1].  A cursor
1.5      rin  * used to incrementally tokenize this string is returned in
1.1  abhinav  * *ppCursor.
1.1  abhinav  */
1.5      rin static int
1.1  abhinav aproposPorterOpen(
1.1  abhinav     sqlite3_tokenizer * pTokenizer,	/* The tokenizer */
1.1  abhinav     const char *zInput, int nInput,	/* String to be tokenized */
1.1  abhinav     sqlite3_tokenizer_cursor ** ppCursor	/* OUT: Tokenization cursor */
1.1  abhinav )
1.1  abhinav {
1.1  abhinav 	custom_apropos_tokenizer_cursor *c;
1.1  abhinav
1.1  abhinav 	c = calloc(1, sizeof(*c));
1.1  abhinav 	if (c == NULL)
1.1  abhinav 		return SQLITE_NOMEM;
1.1  abhinav
1.1  abhinav 	c->zInput = zInput;
1.1  abhinav 	if (zInput != 0) {
1.1  abhinav 		if (nInput < 0)
1.1  abhinav 			c->nInput = strlen(zInput);
1.1  abhinav 		else
1.1  abhinav 			c->nInput = nInput;
1.1  abhinav 	}
1.1  abhinav
1.1  abhinav 	*ppCursor = &c->base;
1.1  abhinav 	return SQLITE_OK;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Close a tokenization cursor previously opened by a call to
1.1  abhinav  * aproposPorterOpen() above.
1.1  abhinav  */
1.5      rin static int
1.1  abhinav aproposPorterClose(sqlite3_tokenizer_cursor *pCursor)
1.1  abhinav {
1.1  abhinav 	custom_apropos_tokenizer_cursor *c = (custom_apropos_tokenizer_cursor *) pCursor;
1.1  abhinav 	free(c->zToken);
1.1  abhinav 	free(c);
1.1  abhinav 	return SQLITE_OK;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Vowel or consonant
1.1  abhinav  */
1.1  abhinav static const char cType[] = {
1.1  abhinav 	0, 1, 1, 1, 0, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0, 1, 1, 1, 1, 1, 0,
1.1  abhinav 	1, 1, 1, 2, 1
1.1  abhinav };
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * isConsonant() and isVowel() determine if their first character in
1.1  abhinav  * the string they point to is a consonant or a vowel, according
1.5      rin  * to Porter ruls.
1.1  abhinav  *
1.1  abhinav  * A consonate is any letter other than 'a', 'e', 'i', 'o', or 'u'.
1.1  abhinav  * 'Y' is a consonant unless it follows another consonant,
1.1  abhinav  * in which case it is a vowel.
1.1  abhinav  *
1.1  abhinav  * In these routine, the letters are in reverse order.  So the 'y' rule
1.1  abhinav  * is that 'y' is a consonant unless it is followed by another
1.1  abhinav  * consonent.
1.1  abhinav  */
1.1  abhinav static int isVowel(const char*);
1.1  abhinav
1.5      rin static int
1.1  abhinav isConsonant(const char *z)
1.1  abhinav {
1.1  abhinav 	int j;
1.1  abhinav 	char x = *z;
1.1  abhinav 	if (x == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	assert(x >= 'a' && x <= 'z');
1.1  abhinav 	j = cType[x - 'a'];
1.1  abhinav 	if (j < 2)
1.1  abhinav 		return j;
1.1  abhinav 	return z[1] == 0 || isVowel(z + 1);
1.1  abhinav }
1.1  abhinav
1.5      rin static int
1.1  abhinav isVowel(const char *z)
1.1  abhinav {
1.1  abhinav 	int j;
1.1  abhinav 	char x = *z;
1.1  abhinav 	if (x == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	assert(x >= 'a' && x <= 'z');
1.1  abhinav 	j = cType[x - 'a'];
1.1  abhinav 	if (j < 2)
1.1  abhinav 		return 1 - j;
1.1  abhinav 	return isConsonant(z + 1);
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Let any sequence of one or more vowels be represented by V and let
1.1  abhinav  * C be sequence of one or more consonants.  Then every word can be
1.1  abhinav  * represented as:
1.1  abhinav  *
1.1  abhinav  *           [C] (VC){m} [V]
1.1  abhinav  *
1.1  abhinav  * In prose:  A word is an optional consonant followed by zero or
1.1  abhinav  * vowel-consonant pairs followed by an optional vowel.  "m" is the
1.1  abhinav  * number of vowel consonant pairs.  This routine computes the value
1.1  abhinav  * of m for the first i bytes of a word.
1.1  abhinav  *
1.1  abhinav  * Return true if the m-value for z is 1 or more.  In other words,
1.1  abhinav  * return true if z contains at least one vowel that is followed
1.1  abhinav  * by a consonant.
1.1  abhinav  *
1.1  abhinav  * In this routine z[] is in reverse order.  So we are really looking
1.1  abhinav  * for an instance of a consonant followed by a vowel.
1.1  abhinav  */
1.5      rin static int
1.1  abhinav m_gt_0(const char *z)
1.1  abhinav {
1.1  abhinav 	while (isVowel(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	return *z != 0;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /* Like mgt0 above except we are looking for a value of m which is
1.1  abhinav  * exactly 1
1.1  abhinav  */
1.5      rin static int
1.1  abhinav m_eq_1(const char *z)
1.1  abhinav {
1.1  abhinav 	while (isVowel(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isVowel(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 1;
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	return *z == 0;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /* Like mgt0 above except we are looking for a value of m>1 instead
1.1  abhinav  * or m>0
1.1  abhinav  */
1.5      rin static int
1.1  abhinav m_gt_1(const char *z)
1.1  abhinav {
1.1  abhinav 	while (isVowel(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isVowel(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	if (*z == 0)
1.1  abhinav 		return 0;
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	return *z != 0;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Return TRUE if there is a vowel anywhere within z[0..n-1]
1.1  abhinav  */
1.5      rin static int
1.1  abhinav hasVowel(const char *z)
1.1  abhinav {
1.1  abhinav 	while (isConsonant(z)) {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	return *z != 0;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Return TRUE if the word ends in a double consonant.
1.1  abhinav  *
1.1  abhinav  * The text is reversed here. So we are really looking at
1.1  abhinav  * the first two characters of z[].
1.1  abhinav  */
1.5      rin static int
1.1  abhinav doubleConsonant(const char *z)
1.1  abhinav {
1.1  abhinav 	return isConsonant(z) && z[0] == z[1];
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Return TRUE if the word ends with three letters which
1.1  abhinav  * are consonant-vowel-consonent and where the final consonant
1.1  abhinav  * is not 'w', 'x', or 'y'.
1.1  abhinav  *
1.1  abhinav  * The word is reversed here.  So we are really checking the
1.1  abhinav  * first three letters and the first one cannot be in [wxy].
1.1  abhinav  */
1.5      rin static int
1.1  abhinav star_oh(const char *z)
1.1  abhinav {
1.1  abhinav 	return isConsonant(z) &&
1.1  abhinav 	    z[0] != 'w' && z[0] != 'x' && z[0] != 'y' &&
1.1  abhinav 	    isVowel(z + 1) &&
1.1  abhinav 	    isConsonant(z + 2);
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * If the word ends with zFrom and xCond() is true for the stem
1.5      rin  * of the word that precedes the zFrom ending, then change the
1.1  abhinav  * ending to zTo.
1.1  abhinav  *
1.1  abhinav  * The input word *pz and zFrom are both in reverse order.  zTo
1.5      rin  * is in normal order.
1.1  abhinav  *
1.1  abhinav  * Return TRUE if zFrom matches.  Return FALSE if zFrom does not
1.1  abhinav  * match.  Not that TRUE is returned even if xCond() fails and
1.1  abhinav  * no substitution occurs.
1.1  abhinav  */
1.5      rin static int
1.1  abhinav stem(
1.1  abhinav     char **pz,			/* The word being stemmed (Reversed) */
1.1  abhinav     const char *zFrom,		/* If the ending matches this... (Reversed) */
1.1  abhinav     const char *zTo,		/* ... change the ending to this (not reversed) */
1.1  abhinav     int (*xCond) (const char *)	/* Condition that must be true */
1.1  abhinav )
1.1  abhinav {
1.1  abhinav 	char *z = *pz;
1.1  abhinav 	while (*zFrom && *zFrom == *z) {
1.1  abhinav 		z++;
1.1  abhinav 		zFrom++;
1.1  abhinav 	}
1.1  abhinav 	if (*zFrom != 0)
1.1  abhinav 		return 0;
1.1  abhinav 	if (xCond && !xCond(z))
1.1  abhinav 		return 1;
1.1  abhinav 	while (*zTo) {
1.1  abhinav 		*(--z) = *(zTo++);
1.1  abhinav 	}
1.1  abhinav 	*pz = z;
1.1  abhinav 	return 1;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * This is the fallback stemmer used when the porter stemmer is
1.1  abhinav  * inappropriate.  The input word is copied into the output with
1.1  abhinav  * US-ASCII case folding.  If the input word is too long (more
1.1  abhinav  * than 20 bytes if it contains no digits or more than 6 bytes if
1.1  abhinav  * it contains digits) then word is truncated to 20 or 6 bytes
1.1  abhinav  * by taking 10 or 3 bytes from the beginning and end.
1.1  abhinav  */
1.5      rin static void
1.1  abhinav copy_stemmer(const char *zIn, size_t nIn, char *zOut, size_t *pnOut)
1.1  abhinav {
1.1  abhinav 	size_t i, mx, j;
1.1  abhinav 	int hasDigit = 0;
1.1  abhinav 	for (i = 0; i < nIn; i++) {
1.1  abhinav 		char c = zIn[i];
1.1  abhinav 		if (c >= 'A' && c <= 'Z') {
1.1  abhinav 			zOut[i] = c - 'A' + 'a';
1.1  abhinav 		} else {
1.1  abhinav 			if (c >= '0' && c <= '9')
1.1  abhinav 				hasDigit = 1;
1.1  abhinav 			zOut[i] = c;
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	mx = hasDigit ? 3 : 10;
1.1  abhinav 	if (nIn > mx * 2) {
1.1  abhinav 		for (j = mx, i = nIn - mx; i < nIn; i++, j++) {
1.1  abhinav 			zOut[j] = zOut[i];
1.1  abhinav 		}
1.1  abhinav 		i = j;
1.1  abhinav 	}
1.1  abhinav 	zOut[i] = 0;
1.1  abhinav 	*pnOut = i;
1.1  abhinav }
1.1  abhinav
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Stem the input word zIn[0..nIn-1].  Store the output in zOut.
1.1  abhinav  * zOut is at least big enough to hold nIn bytes.  Write the actual
1.1  abhinav  * size of the output word (exclusive of the '\0' terminator) into *pnOut.
1.1  abhinav  *
1.1  abhinav  * Any upper-case characters in the US-ASCII character set ([A-Z])
1.1  abhinav  * are converted to lower case.  Upper-case UTF characters are
1.1  abhinav  * unchanged.
1.1  abhinav  *
1.1  abhinav  * Words that are longer than about 20 bytes are stemmed by retaining
1.1  abhinav  * a few bytes from the beginning and the end of the word.  If the
1.1  abhinav  * word contains digits, 3 bytes are taken from the beginning and
1.1  abhinav  * 3 bytes from the end.  For long words without digits, 10 bytes
1.1  abhinav  * are taken from each end.  US-ASCII case folding still applies.
1.5      rin  *
1.5      rin  * If the input word contains not digits but does characters not
1.5      rin  * in [a-zA-Z] then no stemming is attempted and this routine just
1.1  abhinav  * copies the input into the input into the output with US-ASCII
1.1  abhinav  * case folding.
1.1  abhinav  *
1.1  abhinav  * Stemming never increases the length of the word.  So there is
1.1  abhinav  * no chance of overflowing the zOut buffer.
1.1  abhinav  */
1.5      rin static void
1.1  abhinav porter_stemmer(const char *zIn, size_t nIn, char *zOut, size_t *pnOut)
1.1  abhinav {
1.1  abhinav 	size_t i, j;
1.1  abhinav 	char zReverse[28];
1.1  abhinav 	char *z, *z2;
1.1  abhinav 	if (nIn < 3 || nIn >= sizeof(zReverse) - 7) {
1.1  abhinav 		/* The word is too big or too small for the porter stemmer.
1.1  abhinav 		 * Fallback to the copy stemmer
1.1  abhinav 		 */
1.1  abhinav 		copy_stemmer(zIn, nIn, zOut, pnOut);
1.1  abhinav 		return;
1.1  abhinav 	}
1.1  abhinav
1.1  abhinav 	for (i = 0, j = sizeof(zReverse) - 6; i < nIn; i++, j--) {
1.1  abhinav 		char c = zIn[i];
1.1  abhinav 		if (c >= 'A' && c <= 'Z') {
1.1  abhinav 			zReverse[j] = c + 'a' - 'A';
1.1  abhinav 		} else if (c >= 'a' && c <= 'z') {
1.1  abhinav 			zReverse[j] = c;
1.1  abhinav 		} else {
1.1  abhinav 			/* The use of a character not in [a-zA-Z] means that
1.1  abhinav 			 * we fallback * to the copy stemmer
1.1  abhinav 			 */
1.1  abhinav 			copy_stemmer(zIn, nIn, zOut, pnOut);
1.1  abhinav 			return;
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	memset(&zReverse[sizeof(zReverse) - 5], 0, 5);
1.1  abhinav 	z = &zReverse[j + 1];
1.1  abhinav
1.1  abhinav
1.1  abhinav 	/* Step 1a */
1.1  abhinav 	if (z[0] == 's') {
1.1  abhinav 		if (
1.1  abhinav 		    !stem(&z, "sess", "ss", 0) &&
1.1  abhinav 		    !stem(&z, "sei", "i", 0) &&
1.1  abhinav 		    !stem(&z, "ss", "ss", 0)
1.1  abhinav 		    ) {
1.1  abhinav 			z++;
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	/* Step 1b */
1.1  abhinav 	z2 = z;
1.1  abhinav 	if (stem(&z, "dee", "ee", m_gt_0)) {
1.1  abhinav 		/* Do nothing.  The work was all in the test */
1.1  abhinav 	} else if (
1.1  abhinav 		    (stem(&z, "gni", "", hasVowel) || stem(&z, "de", "", hasVowel))
1.1  abhinav 		    && z != z2
1.1  abhinav 	    ) {
1.1  abhinav 		if (stem(&z, "ta", "ate", 0) ||
1.1  abhinav 		    stem(&z, "lb", "ble", 0) ||
1.1  abhinav 		    stem(&z, "zi", "ize", 0)) {
1.1  abhinav 			/* Do nothing.  The work was all in the test */
1.1  abhinav 		} else if (doubleConsonant(z) && (*z != 'l' && *z != 's' && *z != 'z')) {
1.1  abhinav 			z++;
1.1  abhinav 		} else if (m_eq_1(z) && star_oh(z)) {
1.1  abhinav 			*(--z) = 'e';
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	/* Step 1c */
1.1  abhinav 	if (z[0] == 'y' && hasVowel(z + 1)) {
1.1  abhinav 		z[0] = 'i';
1.1  abhinav 	}
1.1  abhinav 	/* Step 2 */
1.1  abhinav 	switch (z[1]) {
1.1  abhinav 	case 'a':
1.1  abhinav 		if (!stem(&z, "lanoita", "ate", m_gt_0)) {
1.1  abhinav 			stem(&z, "lanoit", "tion", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'c':
1.1  abhinav 		if (!stem(&z, "icne", "ence", m_gt_0)) {
1.1  abhinav 			stem(&z, "icna", "ance", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'e':
1.1  abhinav 		stem(&z, "rezi", "ize", m_gt_0);
1.1  abhinav 		break;
1.1  abhinav 	case 'g':
1.1  abhinav 		stem(&z, "igol", "log", m_gt_0);
1.1  abhinav 		break;
1.1  abhinav 	case 'l':
1.1  abhinav 		if (!stem(&z, "ilb", "ble", m_gt_0)
1.1  abhinav 		    && !stem(&z, "illa", "al", m_gt_0)
1.1  abhinav 		    && !stem(&z, "iltne", "ent", m_gt_0)
1.1  abhinav 		    && !stem(&z, "ile", "e", m_gt_0)
1.1  abhinav 		    ) {
1.1  abhinav 			stem(&z, "ilsuo", "ous", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'o':
1.1  abhinav 		if (!stem(&z, "noitazi", "ize", m_gt_0)
1.1  abhinav 		    && !stem(&z, "noita", "ate", m_gt_0)
1.1  abhinav 		    ) {
1.1  abhinav 			stem(&z, "rota", "ate", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 's':
1.1  abhinav 		if (!stem(&z, "msila", "al", m_gt_0)
1.1  abhinav 		    && !stem(&z, "ssenevi", "ive", m_gt_0)
1.1  abhinav 		    && !stem(&z, "ssenluf", "ful", m_gt_0)
1.1  abhinav 		    ) {
1.1  abhinav 			stem(&z, "ssensuo", "ous", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 't':
1.1  abhinav 		if (!stem(&z, "itila", "al", m_gt_0)
1.1  abhinav 		    && !stem(&z, "itivi", "ive", m_gt_0)
1.1  abhinav 		    ) {
1.1  abhinav 			stem(&z, "itilib", "ble", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	}
1.1  abhinav
1.1  abhinav 	/* Step 3 */
1.1  abhinav 	switch (z[0]) {
1.1  abhinav 	case 'e':
1.1  abhinav 		if (!stem(&z, "etaci", "ic", m_gt_0)
1.1  abhinav 		    && !stem(&z, "evita", "", m_gt_0)
1.1  abhinav 		    ) {
1.1  abhinav 			stem(&z, "ezila", "al", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'i':
1.1  abhinav 		stem(&z, "itici", "ic", m_gt_0);
1.1  abhinav 		break;
1.1  abhinav 	case 'l':
1.1  abhinav 		if (!stem(&z, "laci", "ic", m_gt_0)) {
1.1  abhinav 			stem(&z, "luf", "", m_gt_0);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 's':
1.1  abhinav 		stem(&z, "ssen", "", m_gt_0);
1.1  abhinav 		break;
1.1  abhinav 	}
1.1  abhinav
1.1  abhinav 	/* Step 4 */
1.1  abhinav 	switch (z[1]) {
1.1  abhinav 	case 'a':
1.1  abhinav 		if (z[0] == 'l' && m_gt_1(z + 2)) {
1.1  abhinav 			z += 2;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'c':
1.1  abhinav 		if (z[0] == 'e' && z[2] == 'n' && (z[3] == 'a' || z[3] == 'e') && m_gt_1(z + 4)) {
1.1  abhinav 			z += 4;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'e':
1.1  abhinav 		if (z[0] == 'r' && m_gt_1(z + 2)) {
1.1  abhinav 			z += 2;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'i':
1.1  abhinav 		if (z[0] == 'c' && m_gt_1(z + 2)) {
1.1  abhinav 			z += 2;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'l':
1.1  abhinav 		if (z[0] == 'e' && z[2] == 'b' && (z[3] == 'a' || z[3] == 'i') && m_gt_1(z + 4)) {
1.1  abhinav 			z += 4;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'n':
1.1  abhinav 		if (z[0] == 't') {
1.1  abhinav 			if (z[2] == 'a') {
1.1  abhinav 				if (m_gt_1(z + 3)) {
1.1  abhinav 					z += 3;
1.1  abhinav 				}
1.1  abhinav 			} else if (z[2] == 'e') {
1.1  abhinav 				if (!stem(&z, "tneme", "", m_gt_1)
1.1  abhinav 				    && !stem(&z, "tnem", "", m_gt_1)
1.1  abhinav 				    ) {
1.1  abhinav 					stem(&z, "tne", "", m_gt_1);
1.1  abhinav 				}
1.1  abhinav 			}
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'o':
1.1  abhinav 		if (z[0] == 'u') {
1.1  abhinav 			if (m_gt_1(z + 2)) {
1.1  abhinav 				z += 2;
1.1  abhinav 			}
1.1  abhinav 		} else if (z[3] == 's' || z[3] == 't') {
1.1  abhinav 			stem(&z, "noi", "", m_gt_1);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 's':
1.1  abhinav 		if (z[0] == 'm' && z[2] == 'i' && m_gt_1(z + 3)) {
1.1  abhinav 			z += 3;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 't':
1.1  abhinav 		if (!stem(&z, "eta", "", m_gt_1)) {
1.1  abhinav 			stem(&z, "iti", "", m_gt_1);
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'u':
1.1  abhinav 		if (z[0] == 's' && z[2] == 'o' && m_gt_1(z + 3)) {
1.1  abhinav 			z += 3;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	case 'v':
1.1  abhinav 	case 'z':
1.1  abhinav 		if (z[0] == 'e' && z[2] == 'i' && m_gt_1(z + 3)) {
1.1  abhinav 			z += 3;
1.1  abhinav 		}
1.1  abhinav 		break;
1.1  abhinav 	}
1.1  abhinav
1.1  abhinav 	/* Step 5a */
1.1  abhinav 	if (z[0] == 'e') {
1.1  abhinav 		if (m_gt_1(z + 1)) {
1.1  abhinav 			z++;
1.1  abhinav 		} else if (m_eq_1(z + 1) && !star_oh(z + 1)) {
1.1  abhinav 			z++;
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	/* Step 5b */
1.1  abhinav 	if (m_gt_1(z) && z[0] == 'l' && z[1] == 'l') {
1.1  abhinav 		z++;
1.1  abhinav 	}
1.1  abhinav 	/* z[] is now the stemmed word in reverse order.  Flip it back
1.1  abhinav 	 * around into forward order and return.
1.1  abhinav 	 */
1.1  abhinav 	*pnOut = i = strlen(z);
1.1  abhinav 	zOut[i] = 0;
1.1  abhinav 	while (*z) {
1.1  abhinav 		zOut[--i] = *(z++);
1.1  abhinav 	}
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Based on whether the input word is in the nostem list or not
1.1  abhinav  * call porter stemmer to stem it, or call copy_stemmer to keep it
1.5      rin  * as it is (copy_stemmer converts simply converts it to lower case)
1.1  abhinav  * Returns  SQLITE_OK if stemming is successful, an error code for
1.1  abhinav  * any errors
1.1  abhinav  */
1.1  abhinav static int
1.1  abhinav do_stem(const char *zIn, size_t nIn, char *zOut, size_t *pnOut)
1.1  abhinav {
1.1  abhinav 	/* Before looking up the word in the hash table, convert it to lower-case */
1.1  abhinav 	char *dupword = malloc(nIn);
1.1  abhinav 	if (dupword == NULL)
1.1  abhinav 		return SQLITE_NOMEM;
1.1  abhinav
1.1  abhinav 	for (size_t i = 0; i < nIn; i++)
1.1  abhinav 		dupword[i] = tolower((unsigned char) zIn[i]);
1.1  abhinav
1.1  abhinav 	size_t idx = nostem_hash(dupword, nIn);
1.1  abhinav 	if (strncmp(nostem[idx], dupword, nIn) == 0 && nostem[idx][nIn] == 0)
1.1  abhinav 		copy_stemmer(zIn, nIn, zOut, pnOut);
1.1  abhinav 	else
1.1  abhinav 		porter_stemmer(zIn, nIn, zOut, pnOut);
1.1  abhinav
1.1  abhinav 	free(dupword);
1.1  abhinav 	return SQLITE_OK;
1.1  abhinav }
1.1  abhinav
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Characters that can be part of a token.  We assume any character
1.1  abhinav  * whose value is greater than 0x80 (any UTF character) can be
1.1  abhinav  * part of a token.  In other words, delimiters all must have
1.1  abhinav  * values of 0x7f or lower.
1.1  abhinav  */
1.1  abhinav static const char porterIdChar[] = {
1.1  abhinav /* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
1.1  abhinav 	1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0,	/* 3x */
1.1  abhinav 	0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,	/* 4x */
1.1  abhinav 	1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0,	/* 5x */
1.1  abhinav 	0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,	/* 6x */
1.1  abhinav 	1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0,	/* 7x */
1.1  abhinav };
1.1  abhinav
1.1  abhinav #define isDelim(C) (((ch=C)&0x80)==0 && (ch<0x30 || !porterIdChar[ch-0x30]))
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Extract the next token from a tokenization cursor.  The cursor must
1.1  abhinav  * have been opened by a prior call to aproposPorterOpen().
1.1  abhinav  */
1.5      rin static int
1.1  abhinav aproposPorterNext(
1.1  abhinav     sqlite3_tokenizer_cursor *pCursor,	/* Cursor returned by aproposPorterOpen */
1.1  abhinav     const char **pzToken,	/* OUT: *pzToken is the token text */
1.1  abhinav     int *pnBytes,		/* OUT: Number of bytes in token */
1.1  abhinav     int *piStartOffset,		/* OUT: Starting offset of token */
1.1  abhinav     int *piEndOffset,		/* OUT: Ending offset of token */
1.1  abhinav     int *piPosition		/* OUT: Position integer of token */
1.1  abhinav )
1.1  abhinav {
1.1  abhinav 	custom_apropos_tokenizer_cursor *c = (custom_apropos_tokenizer_cursor *) pCursor;
1.1  abhinav 	const char *z = c->zInput;
1.1  abhinav
1.1  abhinav 	while (c->iOffset < c->nInput) {
1.1  abhinav 		size_t iStartOffset, ch;
1.1  abhinav
1.1  abhinav 		/* Scan past delimiter characters */
1.1  abhinav 		while (c->iOffset < c->nInput && isDelim(z[c->iOffset])) {
1.1  abhinav 			c->iOffset++;
1.1  abhinav 		}
1.1  abhinav
1.1  abhinav 		/* Count non-delimiter characters. */
1.1  abhinav 		iStartOffset = c->iOffset;
1.1  abhinav 		while (c->iOffset < c->nInput && !isDelim(z[c->iOffset])) {
1.1  abhinav 			c->iOffset++;
1.1  abhinav 		}
1.1  abhinav
1.1  abhinav 		if (c->iOffset > iStartOffset) {
1.1  abhinav 			size_t n = c->iOffset - iStartOffset;
1.1  abhinav 			if (n > c->nAllocated) {
1.1  abhinav 				char *pNew;
1.1  abhinav 				c->nAllocated = n + 20;
1.1  abhinav 				pNew = realloc(c->zToken, c->nAllocated);
1.1  abhinav 				if (!pNew)
1.1  abhinav 					return SQLITE_NOMEM;
1.1  abhinav 				c->zToken = pNew;
1.1  abhinav 			}
1.5      rin
1.5      rin 			size_t temp;
1.2  abhinav 			int stemStatus = do_stem(&z[iStartOffset], n, c->zToken, &temp);
1.1  abhinav 			if (stemStatus != SQLITE_OK)
1.1  abhinav 				return stemStatus;
1.6      tnn 			*pnBytes = temp;
1.1  abhinav
1.1  abhinav 			*pzToken = c->zToken;
1.1  abhinav 			*piStartOffset = iStartOffset;
1.1  abhinav 			*piEndOffset = c->iOffset;
1.1  abhinav 			*piPosition = c->iToken++;
1.1  abhinav 			return SQLITE_OK;
1.1  abhinav 		}
1.1  abhinav 	}
1.1  abhinav 	return SQLITE_DONE;
1.1  abhinav }
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * The set of routines that implement the porter-stemmer tokenizer
1.1  abhinav  */
1.1  abhinav static const sqlite3_tokenizer_module aproposPorterTokenizerModule = {
1.1  abhinav 	0,
1.1  abhinav 	aproposPorterCreate,
1.1  abhinav 	aproposPorterDestroy,
1.1  abhinav 	aproposPorterOpen,
1.1  abhinav 	aproposPorterClose,
1.1  abhinav 	aproposPorterNext,
1.1  abhinav 	0
1.1  abhinav };
1.1  abhinav
1.1  abhinav /*
1.1  abhinav  * Allocate a new porter tokenizer.  Return a pointer to the new
1.1  abhinav  * tokenizer in *ppModule
1.1  abhinav  */
1.5      rin void
1.1  abhinav get_custom_apropos_tokenizer(sqlite3_tokenizer_module const ** ppModule)
1.1  abhinav {
1.1  abhinav 	*ppModule = &aproposPorterTokenizerModule;
1.1  abhinav }