Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 44 additions & 5 deletions contrib/amcheck/expected/check_btree.out
Original file line number Diff line number Diff line change
Expand Up @@ -114,23 +114,62 @@ WHERE relation = ANY(ARRAY['bttest_a', 'bttest_a_idx', 'bttest_b', 'bttest_b_idx

COMMIT;
--
-- Check that index expressions and predicates are run as the table's owner
-- Check that index expressions and predicates are run as the table's owner,
-- with empty search_path
--
TRUNCATE bttest_a;
INSERT INTO bttest_a SELECT * FROM generate_series(1, 1000);
ALTER TABLE bttest_a OWNER TO regress_bttest_role;
-- A dummy index function checking current_user
CREATE FUNCTION ifun(int8) RETURNS int8 AS $$
BEGIN
IF current_user <> 'regress_bttest_role'
THEN RAISE EXCEPTION 'ifun(%s) called by %s', $1, current_user;
END IF;
-- ASSERT is unavailable in GPDB (based on PG 9.4), use IF/RAISE instead
IF current_setting('search_path') LIKE '%preempt%' THEN
RAISE EXCEPTION '%', format('ifun(%s) called with current_schemas %s, search_path %s',
$1, current_schemas(true), current_setting('search_path'));
END IF;
IF "current_user"() <> 'regress_bttest_role' THEN
RAISE EXCEPTION '%', format('ifun(%s) called by %s', $1, current_user);
END IF;
RETURN $1;
END;
$$ LANGUAGE plpgsql IMMUTABLE;
CREATE INDEX bttest_a_expr_idx ON bttest_a ((ifun(id) + ifun(0)))
WHERE ifun(id + 10) > ifun(10);
SELECT bt_index_check('bttest_a_expr_idx');
BEGIN;
SET LOCAL check_function_bodies = off;
CREATE SCHEMA preempt;
GRANT USAGE ON SCHEMA preempt TO regress_bttest_role;
SET LOCAL search_path = preempt, pg_catalog, public;
CREATE FUNCTION "current_user"() RETURNS name AS $$
broken
$$ LANGUAGE sql STABLE STRICT;
SELECT bt_index_check('bttest_a_expr_idx', true);
bt_index_check
----------------

(1 row)

ROLLBACK;
-- Check support of both 1B and 4B header sizes of short varlena datum
CREATE TABLE varlena_bug (v text);
NOTICE: Table doesn't have 'DISTRIBUTED BY' clause -- Using column named 'v' as the Greenplum Database data distribution key for this table.
HINT: The 'DISTRIBUTED BY' clause determines the distribution of data. Make sure column(s) chosen are the optimal data distribution key to minimize skew.
ALTER TABLE varlena_bug ALTER column v SET storage plain;
INSERT INTO varlena_bug VALUES ('x');
COPY varlena_bug from stdin;
CREATE INDEX varlena_bug_idx on varlena_bug(v);
SELECT bt_index_check('varlena_bug_idx', true);
bt_index_check
----------------

(1 row)

-- Also check that we compress varlena values, which were previously stored
-- uncompressed in index.
INSERT INTO varlena_bug VALUES (repeat('Test', 250));
ALTER TABLE varlena_bug ALTER COLUMN v SET STORAGE extended;
SELECT bt_index_check('varlena_bug_idx', true);
bt_index_check
----------------

Expand Down
40 changes: 35 additions & 5 deletions contrib/amcheck/sql/check_btree.sql
Original file line number Diff line number Diff line change
Expand Up @@ -61,25 +61,55 @@ WHERE relation = ANY(ARRAY['bttest_a', 'bttest_a_idx', 'bttest_b', 'bttest_b_idx
COMMIT;

--
-- Check that index expressions and predicates are run as the table's owner
-- Check that index expressions and predicates are run as the table's owner,
-- with empty search_path
--
TRUNCATE bttest_a;
INSERT INTO bttest_a SELECT * FROM generate_series(1, 1000);
ALTER TABLE bttest_a OWNER TO regress_bttest_role;
-- A dummy index function checking current_user
CREATE FUNCTION ifun(int8) RETURNS int8 AS $$
BEGIN
IF current_user <> 'regress_bttest_role'
THEN RAISE EXCEPTION 'ifun(%s) called by %s', $1, current_user;
END IF;
-- ASSERT is unavailable in GPDB (based on PG 9.4), use IF/RAISE instead
IF current_setting('search_path') LIKE '%preempt%' THEN
RAISE EXCEPTION '%', format('ifun(%s) called with current_schemas %s, search_path %s',
$1, current_schemas(true), current_setting('search_path'));
END IF;
IF "current_user"() <> 'regress_bttest_role' THEN
RAISE EXCEPTION '%', format('ifun(%s) called by %s', $1, current_user);
END IF;
RETURN $1;
END;
$$ LANGUAGE plpgsql IMMUTABLE;

CREATE INDEX bttest_a_expr_idx ON bttest_a ((ifun(id) + ifun(0)))
WHERE ifun(id + 10) > ifun(10);
BEGIN;
SET LOCAL check_function_bodies = off;
CREATE SCHEMA preempt;
GRANT USAGE ON SCHEMA preempt TO regress_bttest_role;
SET LOCAL search_path = preempt, pg_catalog, public;
CREATE FUNCTION "current_user"() RETURNS name AS $$
broken
$$ LANGUAGE sql STABLE STRICT;
SELECT bt_index_check('bttest_a_expr_idx', true);
ROLLBACK;

SELECT bt_index_check('bttest_a_expr_idx');
-- Check support of both 1B and 4B header sizes of short varlena datum
CREATE TABLE varlena_bug (v text);
ALTER TABLE varlena_bug ALTER column v SET storage plain;
INSERT INTO varlena_bug VALUES ('x');
COPY varlena_bug from stdin;
x
\.
CREATE INDEX varlena_bug_idx on varlena_bug(v);
SELECT bt_index_check('varlena_bug_idx', true);

-- Also check that we compress varlena values, which were previously stored
-- uncompressed in index.
INSERT INTO varlena_bug VALUES (repeat('Test', 250));
ALTER TABLE varlena_bug ALTER COLUMN v SET STORAGE extended;
SELECT bt_index_check('varlena_bug_idx', true);

-- cleanup
DROP TABLE bttest_a;
Expand Down
2 changes: 2 additions & 0 deletions contrib/amcheck/verify_nbtree.c
Original file line number Diff line number Diff line change
Expand Up @@ -232,6 +232,8 @@ bt_index_check_internal(Oid indrelid, bool parentcheck, bool heapallindexed)
SetUserIdAndSecContext(heaprel->rd_rel->relowner,
save_sec_context | SECURITY_RESTRICTED_OPERATION);
save_nestlevel = NewGUCNestLevel();
set_config_option("search_path", "pg_catalog, pg_temp", PGC_USERSET,
PGC_S_SESSION, GUC_ACTION_SAVE, true, 0);
}
else
{
Expand Down
14 changes: 14 additions & 0 deletions contrib/fuzzystrmatch/fuzzystrmatch.c
Original file line number Diff line number Diff line change
Expand Up @@ -167,6 +167,20 @@ rest_of_char_same(const char *s1, const char *s2, int len)
return true;
}

/*
* Helper function for checking return value of Levenshtein distance functions.
* We calculate it as an int64, but the distance functions return an int32.
*/
static inline int
levenshtein_result(int64 res)
{
if (unlikely(res < PG_INT32_MIN || res > PG_INT32_MAX))
ereport(ERROR,
(errcode(ERRCODE_NUMERIC_VALUE_OUT_OF_RANGE),
errmsg("levenshtein distance out of range")));
return res;
}

#include "levenshtein.c"
#define LEVENSHTEIN_LESS_EQUAL
#include "levenshtein.c"
Expand Down
87 changes: 45 additions & 42 deletions contrib/fuzzystrmatch/levenshtein.c
Original file line number Diff line number Diff line change
Expand Up @@ -76,14 +76,17 @@ levenshtein_internal(text *s, text *t,
n,
s_bytes,
t_bytes;
int *prev;
int *curr;
int64 *prev;
int64 *curr;
int *s_char_len = NULL;
int i,
j;
const char *s_data;
const char *t_data;
const char *y;
int64 ins_c_64 = ins_c;
int64 del_c_64 = del_c;
int64 sub_c_64 = sub_c;

/*
* For levenshtein_less_equal_internal, we have real variables called
Expand Down Expand Up @@ -120,9 +123,9 @@ levenshtein_internal(text *s, text *t,
* into an empty s with m deletions.
*/
if (!m)
return n * ins_c;
return levenshtein_result(n * ins_c_64);
if (!n)
return m * del_c;
return levenshtein_result(m * del_c_64);

/*
* For security concerns, restrict excessive CPU+RAM usage. (This
Expand All @@ -148,20 +151,20 @@ levenshtein_internal(text *s, text *t,
*/
if (max_d >= 0)
{
int min_theo_d; /* Theoretical minimum distance. */
int max_theo_d; /* Theoretical maximum distance. */
int64 min_theo_d; /* Theoretical minimum distance. */
int64 max_theo_d; /* Theoretical maximum distance. */
int net_inserts = n - m;

min_theo_d = net_inserts < 0 ?
-net_inserts * del_c : net_inserts * ins_c;
-net_inserts * del_c_64 : net_inserts * ins_c_64;
if (min_theo_d > max_d)
return max_d + 1;
if (ins_c + del_c < sub_c)
sub_c = ins_c + del_c;
max_theo_d = min_theo_d + sub_c * Min(m, n);
return levenshtein_result((int64) max_d + 1);
if (ins_c_64 + del_c_64 < sub_c_64)
sub_c_64 = ins_c_64 + del_c_64;
max_theo_d = min_theo_d + sub_c_64 * Min(m, n);
if (max_d >= max_theo_d)
max_d = -1;
else if (ins_c + del_c > 0)
else if (ins_c_64 + del_c_64 > 0)
{
/*
* Figure out how much of the first row of the notional matrix we
Expand All @@ -175,12 +178,12 @@ levenshtein_internal(text *s, text *t,
* column n - m. If we do start further right, the best-case
* total cost increases by ins_c + del_c for each move right.
*/
int slack_d = max_d - min_theo_d;
int64 slack_d = max_d - min_theo_d;
int best_column = net_inserts < 0 ? -net_inserts : 0;
int64 tmp;

stop_column = best_column + (slack_d / (ins_c + del_c)) + 1;
if (stop_column > m)
stop_column = m + 1;
tmp = best_column + (slack_d / (ins_c_64 + del_c_64)) + 1;
stop_column = Min(tmp, m + 1);
}
}
#endif
Expand Down Expand Up @@ -212,20 +215,20 @@ levenshtein_internal(text *s, text *t,
++n;

/* Previous and current rows of notional array. */
prev = (int *) palloc(2 * m * sizeof(int));
prev = (int64 *) palloc(2 * m * sizeof(int64));
curr = prev + m;

/*
* To transform the first i characters of s into the first 0 characters of
* t, we must perform i deletions.
*/
for (i = START_COLUMN; i < STOP_COLUMN; i++)
prev[i] = i * del_c;
prev[i] = i * del_c_64;

/* Loop through rows of the notional array */
for (y = t_data, j = 1; j < n; j++)
{
int *temp;
int64 *temp;
const char *x = s_data;
int y_char_len = n != t_bytes + 1 ? pg_mblen(y) : 1;

Expand All @@ -239,7 +242,7 @@ levenshtein_internal(text *s, text *t,
*/
if (stop_column < m)
{
prev[stop_column] = max_d + 1;
prev[stop_column] = (int64) max_d + 1;
++stop_column;
}

Expand All @@ -251,13 +254,13 @@ levenshtein_internal(text *s, text *t,
*/
if (start_column == 0)
{
curr[0] = j * ins_c;
curr[0] = j * ins_c_64;
i = 1;
}
else
i = start_column;
#else
curr[0] = j * ins_c;
curr[0] = j * ins_c_64;
i = 1;
#endif

Expand All @@ -272,9 +275,9 @@ levenshtein_internal(text *s, text *t,
{
for (; i < STOP_COLUMN; i++)
{
int ins;
int del;
int sub;
int64 ins;
int64 del;
int64 sub;
int x_char_len = s_char_len[i - 1];

/*
Expand All @@ -286,14 +289,14 @@ levenshtein_internal(text *s, text *t,
* get past that test, then we compare the lengths and the
* remaining bytes.
*/
ins = prev[i] + ins_c;
del = curr[i - 1] + del_c;
ins = prev[i] + ins_c_64;
del = curr[i - 1] + del_c_64;
if (x[x_char_len - 1] == y[y_char_len - 1]
&& x_char_len == y_char_len &&
(x_char_len == 1 || rest_of_char_same(x, y, x_char_len)))
sub = prev[i - 1];
else
sub = prev[i - 1] + sub_c;
sub = prev[i - 1] + sub_c_64;

/* Take the one with minimum cost. */
curr[i] = Min(ins, del);
Expand All @@ -307,14 +310,14 @@ levenshtein_internal(text *s, text *t,
{
for (; i < STOP_COLUMN; i++)
{
int ins;
int del;
int sub;
int64 ins;
int64 del;
int64 sub;

/* Calculate costs for insertion, deletion, and substitution. */
ins = prev[i] + ins_c;
del = curr[i - 1] + del_c;
sub = prev[i - 1] + ((*x == *y) ? 0 : sub_c);
ins = prev[i] + ins_c_64;
del = curr[i - 1] + del_c_64;
sub = prev[i - 1] + ((*x == *y) ? 0 : sub_c_64);

/* Take the one with minimum cost. */
curr[i] = Min(ins, del);
Expand Down Expand Up @@ -360,8 +363,8 @@ levenshtein_internal(text *s, text *t,
int ii = stop_column - 1;
int net_inserts = ii - zp;

if (prev[ii] + (net_inserts > 0 ? net_inserts * ins_c :
-net_inserts * del_c) <= max_d)
if (prev[ii] + (net_inserts > 0 ? net_inserts * ins_c_64 :
-net_inserts * del_c_64) <= max_d)
break;
stop_column--;
}
Expand All @@ -372,25 +375,25 @@ levenshtein_internal(text *s, text *t,
int net_inserts = start_column - zp;

if (prev[start_column] +
(net_inserts > 0 ? net_inserts * ins_c :
-net_inserts * del_c) <= max_d)
(net_inserts > 0 ? net_inserts * ins_c_64 :
-net_inserts * del_c_64) <= max_d)
break;

/*
* We'll never again update these values, so we must make sure
* there's nothing here that could confuse any future
* iteration of the outer loop.
*/
prev[start_column] = max_d + 1;
curr[start_column] = max_d + 1;
prev[start_column] = (int64) max_d + 1;
curr[start_column] = (int64) max_d + 1;
if (start_column != 0)
s_data += (s_char_len != NULL) ? s_char_len[start_column - 1] : 1;
start_column++;
}

/* If they cross, we're going to exceed the bound. */
if (start_column >= stop_column)
return max_d + 1;
return levenshtein_result((int64) max_d + 1);
}
#endif
}
Expand All @@ -399,5 +402,5 @@ levenshtein_internal(text *s, text *t,
* Because the final value was swapped from the previous row to the
* current row, that's where we'll find it.
*/
return prev[m - 1];
return levenshtein_result(prev[m - 1]);
}
Loading
Loading