]> git.sur5r.net Git - openldap/blobdiff - libraries/liblmdb/mdb.c
ITS#7736 fix regression in ITS#7733 patch
[openldap] / libraries / liblmdb / mdb.c
index 8d2a43c5db492ac9b42407c1a881811c1212222f..9ade333ed951e459ce623b1c1ca0b29169494cf9 100644 (file)
@@ -75,6 +75,7 @@
 #ifndef _WIN32
 #include <pthread.h>
 #ifdef MDB_USE_POSIX_SEM
+# define MDB_USE_HASH          1
 #include <semaphore.h>
 #endif
 #endif
  *     @{
  */
 #ifdef _WIN32
+#define MDB_USE_HASH   1
 #define MDB_PIDLOCK    0
 #define pthread_t      DWORD
 #define pthread_mutex_t        HANDLE
@@ -317,6 +319,9 @@ static txnid_t mdb_debug_start;
         *      The string is printed literally, with no format processing.
         */
 #define DPUTS(arg)     DPRINTF(("%s", arg))
+       /** Debuging output value of a cursor DBI: Negative in a sub-cursor. */
+#define DDBI(mc) \
+       (((mc)->mc_flags & C_SUB) ? -(int)(mc)->mc_dbi : (int)(mc)->mc_dbi)
 /** @} */
 
        /** A default memory page size.
@@ -425,7 +430,8 @@ typedef uint16_t     indx_t;
  *
  *     If #MDB_NOTLS is set, the slot address is not saved in thread-specific data.
  *
- *     No reader table is used if the database is on a read-only filesystem.
+ *     No reader table is used if the database is on a read-only filesystem, or
+ *     if #MDB_NOLOCK is set.
  *
  *     Since the database uses multi-version concurrency control, readers don't
  *     actually need any locking. This table is used to keep track of which
@@ -868,7 +874,7 @@ struct MDB_txn {
  *     @ingroup internal
  * @{
  */
-#define DB_DIRTY       0x01            /**< DB was written in this txn */
+#define DB_DIRTY       0x01            /**< DB was modified or is DUPSORT data */
 #define DB_STALE       0x02            /**< Named-DB record is older than txnID */
 #define DB_NEW         0x04            /**< Named-DB handle opened in this txn */
 #define DB_VALID       0x08            /**< DB handle is valid, see also #MDB_VALID */
@@ -898,10 +904,6 @@ struct MDB_txn {
         *      dirty_list into mt_parent after freeing hidden mt_parent pages.
         */
        unsigned int    mt_dirty_room;
-       /** Tracks which of the two meta pages was used at the start
-        *      of this transaction.
-        */
-       unsigned int    mt_toggle;
 };
 
 /** Enough space for 2^32 nodes with minimum of 2 keys per node. I.e., plenty.
@@ -1576,12 +1578,14 @@ mdb_find_oldest(MDB_txn *txn)
 {
        int i;
        txnid_t mr, oldest = txn->mt_txnid - 1;
-       MDB_reader *r = txn->mt_env->me_txns->mti_readers;
-       for (i = txn->mt_env->me_txns->mti_numreaders; --i >= 0; ) {
-               if (r[i].mr_pid) {
-                       mr = r[i].mr_txnid;
-                       if (oldest > mr)
-                               oldest = mr;
+       if (txn->mt_env->me_txns) {
+               MDB_reader *r = txn->mt_env->me_txns->mti_readers;
+               for (i = txn->mt_env->me_txns->mti_numreaders; --i >= 0; ) {
+                       if (r[i].mr_pid) {
+                               mr = r[i].mr_txnid;
+                               if (oldest > mr)
+                                       oldest = mr;
+                       }
                }
        }
        return oldest;
@@ -1860,7 +1864,6 @@ mdb_page_touch(MDB_cursor *mc)
        MDB_page *mp = mc->mc_pg[mc->mc_top], *np;
        MDB_txn *txn = mc->mc_txn;
        MDB_cursor *m2, *m3;
-       MDB_dbi dbi;
        pgno_t  pgno;
        int rc;
 
@@ -1877,7 +1880,8 @@ mdb_page_touch(MDB_cursor *mc)
                        (rc = mdb_page_alloc(mc, 1, &np)))
                        return rc;
                pgno = np->mp_pgno;
-               DPRINTF(("touched db %u page %"Z"u -> %"Z"u", mc->mc_dbi,mp->mp_pgno,pgno));
+               DPRINTF(("touched db %d page %"Z"u -> %"Z"u", DDBI(mc),
+                       mp->mp_pgno, pgno));
                assert(mp->mp_pgno != pgno);
                mdb_midl_xappend(txn->mt_free_pgs, mp->mp_pgno);
                /* Update the parent page, if any, to point to the new page */
@@ -1923,17 +1927,16 @@ mdb_page_touch(MDB_cursor *mc)
 done:
        /* Adjust cursors pointing to mp */
        mc->mc_pg[mc->mc_top] = np;
-       dbi = mc->mc_dbi;
+       m2 = txn->mt_cursors[mc->mc_dbi];
        if (mc->mc_flags & C_SUB) {
-               dbi--;
-               for (m2 = txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
+               for (; m2; m2=m2->mc_next) {
                        m3 = &m2->mc_xcursor->mx_cursor;
                        if (m3->mc_snum < mc->mc_snum) continue;
                        if (m3->mc_pg[mc->mc_top] == mp)
                                m3->mc_pg[mc->mc_top] = np;
                }
        } else {
-               for (m2 = txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
+               for (; m2; m2=m2->mc_next) {
                        if (m2->mc_snum < mc->mc_snum) continue;
                        if (m2->mc_pg[mc->mc_top] == mp) {
                                m2->mc_pg[mc->mc_top] = np;
@@ -2118,7 +2121,9 @@ static int
 mdb_txn_renew0(MDB_txn *txn)
 {
        MDB_env *env = txn->mt_env;
-       unsigned int i;
+       MDB_txninfo *ti = env->me_txns;
+       MDB_meta *meta;
+       unsigned int i, nr;
        uint16_t x;
        int rc, new_notls = 0;
 
@@ -2127,9 +2132,9 @@ mdb_txn_renew0(MDB_txn *txn)
        txn->mt_dbxs = env->me_dbxs;    /* mostly static anyway */
 
        if (txn->mt_flags & MDB_TXN_RDONLY) {
-               if (!env->me_txns) {
-                       i = mdb_env_pick_meta(env);
-                       txn->mt_txnid = env->me_metas[i]->mm_txnid;
+               if (!ti) {
+                       meta = env->me_metas[ mdb_env_pick_meta(env) ];
+                       txn->mt_txnid = meta->mm_txnid;
                        txn->mt_u.reader = NULL;
                } else {
                        MDB_reader *r = (env->me_flags & MDB_NOTLS) ? txn->mt_u.reader :
@@ -2151,36 +2156,43 @@ mdb_txn_renew0(MDB_txn *txn)
                                }
 
                                LOCK_MUTEX_R(env);
-                               for (i=0; i<env->me_txns->mti_numreaders; i++)
-                                       if (env->me_txns->mti_readers[i].mr_pid == 0)
+                               nr = ti->mti_numreaders;
+                               for (i=0; i<nr; i++)
+                                       if (ti->mti_readers[i].mr_pid == 0)
                                                break;
                                if (i == env->me_maxreaders) {
                                        UNLOCK_MUTEX_R(env);
                                        return MDB_READERS_FULL;
                                }
-                               env->me_txns->mti_readers[i].mr_pid = pid;
-                               env->me_txns->mti_readers[i].mr_tid = tid;
-                               if (i >= env->me_txns->mti_numreaders)
-                                       env->me_txns->mti_numreaders = i+1;
+                               ti->mti_readers[i].mr_pid = pid;
+                               ti->mti_readers[i].mr_tid = tid;
+                               if (i == nr)
+                                       ti->mti_numreaders = ++nr;
                                /* Save numreaders for un-mutexed mdb_env_close() */
-                               env->me_numreaders = env->me_txns->mti_numreaders;
+                               env->me_numreaders = nr;
                                UNLOCK_MUTEX_R(env);
-                               r = &env->me_txns->mti_readers[i];
+
+                               r = &ti->mti_readers[i];
                                new_notls = (env->me_flags & MDB_NOTLS);
                                if (!new_notls && (rc=pthread_setspecific(env->me_txkey, r))) {
                                        r->mr_pid = 0;
                                        return rc;
                                }
                        }
-                       txn->mt_txnid = r->mr_txnid = env->me_txns->mti_txnid;
+                       txn->mt_txnid = r->mr_txnid = ti->mti_txnid;
                        txn->mt_u.reader = r;
+                       meta = env->me_metas[txn->mt_txnid & 1];
                }
-               txn->mt_toggle = txn->mt_txnid & 1;
        } else {
-               LOCK_MUTEX_W(env);
+               if (ti) {
+                       LOCK_MUTEX_W(env);
 
-               txn->mt_txnid = env->me_txns->mti_txnid;
-               txn->mt_toggle = txn->mt_txnid & 1;
+                       txn->mt_txnid = ti->mti_txnid;
+                       meta = env->me_metas[txn->mt_txnid & 1];
+               } else {
+                       meta = env->me_metas[ mdb_env_pick_meta(env) ];
+                       txn->mt_txnid = meta->mm_txnid;
+               }
                txn->mt_txnid++;
 #if MDB_DEBUG
                if (txn->mt_txnid == mdb_debug_start)
@@ -2196,10 +2208,10 @@ mdb_txn_renew0(MDB_txn *txn)
        }
 
        /* Copy the DB info and flags */
-       memcpy(txn->mt_dbs, env->me_metas[txn->mt_toggle]->mm_dbs, 2 * sizeof(MDB_db));
+       memcpy(txn->mt_dbs, meta->mm_dbs, 2 * sizeof(MDB_db));
 
        /* Moved to here to avoid a data race in read TXNs */
-       txn->mt_next_pgno = env->me_metas[txn->mt_toggle]->mm_last_pg+1;
+       txn->mt_next_pgno = meta->mm_last_pg+1;
 
        for (i=2; i<txn->mt_numdbs; i++) {
                x = env->me_dbflags[i];
@@ -2295,7 +2307,6 @@ mdb_txn_begin(MDB_env *env, MDB_txn *parent, unsigned int flags, MDB_txn **ret)
                        return ENOMEM;
                }
                txn->mt_txnid = parent->mt_txnid;
-               txn->mt_toggle = parent->mt_toggle;
                txn->mt_dirty_room = parent->mt_dirty_room;
                txn->mt_u.dirty_list[0].mid = 0;
                txn->mt_spill_pgs = NULL;
@@ -2421,7 +2432,8 @@ mdb_txn_reset0(MDB_txn *txn, const char *act)
 
                env->me_txn = NULL;
                /* The writer mutex was locked in mdb_txn_begin. */
-               UNLOCK_MUTEX_W(env);
+               if (env->me_txns)
+                       UNLOCK_MUTEX_W(env);
        }
 }
 
@@ -2938,7 +2950,8 @@ done:
        env->me_txn = NULL;
        mdb_dbis_update(txn, 1);
 
-       UNLOCK_MUTEX_W(env);
+       if (env->me_txns)
+               UNLOCK_MUTEX_W(env);
        free(txn);
 
        return MDB_SUCCESS;
@@ -3093,7 +3106,7 @@ mdb_env_write_meta(MDB_txn *txn)
        assert(txn != NULL);
        assert(txn->mt_env != NULL);
 
-       toggle = !txn->mt_toggle;
+       toggle = txn->mt_txnid & 1;
        DPRINTF(("writing meta page %d for root page %"Z"u",
                toggle, txn->mt_dbs[MAIN_DBI].md_root));
 
@@ -3184,7 +3197,8 @@ done:
         * readers will get consistent data regardless of how fresh or
         * how stale their view of these values is.
         */
-       env->me_txns->mti_txnid = txn->mt_txnid;
+       if (env->me_txns)
+               env->me_txns->mti_txnid = txn->mt_txnid;
 
        return MDB_SUCCESS;
 }
@@ -3260,7 +3274,7 @@ mdb_env_map(MDB_env *env, void *addr, int newsize)
        int prot = PROT_READ;
        if (flags & MDB_WRITEMAP) {
                prot |= PROT_WRITE;
-               if (newsize && ftruncate(env->me_fd, env->me_mapsize) < 0)
+               if (ftruncate(env->me_fd, env->me_mapsize) < 0)
                        return ErrCode();
        }
        env->me_map = mmap(addr, env->me_mapsize, prot, MAP_SHARED,
@@ -3269,14 +3283,17 @@ mdb_env_map(MDB_env *env, void *addr, int newsize)
                env->me_map = NULL;
                return ErrCode();
        }
-       /* Turn off readahead. It's harmful when the DB is larger than RAM. */
+
+       if (flags & MDB_NORDAHEAD) {
+               /* Turn off readahead. It's harmful when the DB is larger than RAM. */
 #ifdef MADV_RANDOM
-       madvise(env->me_map, env->me_mapsize, MADV_RANDOM);
+               madvise(env->me_map, env->me_mapsize, MADV_RANDOM);
 #else
 #ifdef POSIX_MADV_RANDOM
-       posix_madvise(env->me_map, env->me_mapsize, POSIX_MADV_RANDOM);
+               posix_madvise(env->me_map, env->me_mapsize, POSIX_MADV_RANDOM);
 #endif /* POSIX_MADV_RANDOM */
 #endif /* MADV_RANDOM */
+       }
 #endif /* _WIN32 */
 
        /* Can happen because the address argument to mmap() is just a
@@ -3307,6 +3324,14 @@ mdb_env_set_mapsize(MDB_env *env, size_t size)
                        return EINVAL;
                if (!size)
                        size = env->me_metas[mdb_env_pick_meta(env)]->mm_mapsize;
+               else if (size < env->me_mapsize) {
+                       /* If the configured size is smaller, make sure it's
+                        * still big enough. Silently round up to minimum if not.
+                        */
+                       size_t minsize = (env->me_metas[mdb_env_pick_meta(env)]->mm_last_pg + 1) * env->me_psize;
+                       if (size < minsize)
+                               size = minsize;
+               }
                munmap(env->me_map, env->me_mapsize);
                env->me_mapsize = size;
                old = (env->me_flags & MDB_FIXEDMAP) ? env->me_map : NULL;
@@ -3581,7 +3606,7 @@ mdb_env_excl_lock(MDB_env *env, int *excl)
        return rc;
 }
 
-#if defined(_WIN32) || defined(MDB_USE_POSIX_SEM)
+#ifdef MDB_USE_HASH
 /*
  * hash_64 - 64 bit Fowler/Noll/Vo-0 FNV-1a hash code
  *
@@ -3904,7 +3929,7 @@ fail:
         *      environment and re-opening it with the new flags.
         */
 #define        CHANGEABLE      (MDB_NOSYNC|MDB_NOMETASYNC|MDB_MAPASYNC)
-#define        CHANGELESS      (MDB_FIXEDMAP|MDB_NOSUBDIR|MDB_RDONLY|MDB_WRITEMAP|MDB_NOTLS)
+#define        CHANGELESS      (MDB_FIXEDMAP|MDB_NOSUBDIR|MDB_RDONLY|MDB_WRITEMAP|MDB_NOTLS|MDB_NOLOCK|MDB_NORDAHEAD)
 
 int
 mdb_env_open(MDB_env *env, const char *path, unsigned int flags, mdb_mode_t mode)
@@ -3957,7 +3982,7 @@ mdb_env_open(MDB_env *env, const char *path, unsigned int flags, mdb_mode_t mode
        }
 
        /* For RDONLY, get lockfile after we know datafile exists */
-       if (!F_ISSET(flags, MDB_RDONLY)) {
+       if (!(flags & (MDB_RDONLY|MDB_NOLOCK))) {
                rc = mdb_env_setup_locks(env, lpath, mode, &excl);
                if (rc)
                        goto leave;
@@ -3987,7 +4012,7 @@ mdb_env_open(MDB_env *env, const char *path, unsigned int flags, mdb_mode_t mode
                goto leave;
        }
 
-       if (F_ISSET(flags, MDB_RDONLY)) {
+       if ((flags & (MDB_RDONLY|MDB_NOLOCK)) == MDB_RDONLY) {
                rc = mdb_env_setup_locks(env, lpath, mode, &excl);
                if (rc)
                        goto leave;
@@ -4495,8 +4520,8 @@ mdb_cursor_pop(MDB_cursor *mc)
                if (mc->mc_snum)
                        mc->mc_top--;
 
-               DPRINTF(("popped page %"Z"u off db %u cursor %p", top->mp_pgno,
-                       mc->mc_dbi, (void *) mc));
+               DPRINTF(("popped page %"Z"u off db %d cursor %p", top->mp_pgno,
+                       DDBI(mc), (void *) mc));
        }
 }
 
@@ -4504,8 +4529,8 @@ mdb_cursor_pop(MDB_cursor *mc)
 static int
 mdb_cursor_push(MDB_cursor *mc, MDB_page *mp)
 {
-       DPRINTF(("pushing page %"Z"u on db %u cursor %p", mp->mp_pgno,
-               mc->mc_dbi, (void *) mc));
+       DPRINTF(("pushing page %"Z"u on db %d cursor %p", mp->mp_pgno,
+               DDBI(mc), (void *) mc));
 
        if (mc->mc_snum >= CURSOR_STACK) {
                assert(mc->mc_snum < CURSOR_STACK);
@@ -4738,8 +4763,8 @@ mdb_page_search(MDB_cursor *mc, MDB_val *key, int flags)
        mc->mc_snum = 1;
        mc->mc_top = 0;
 
-       DPRINTF(("db %u root page %"Z"u has flags 0x%X",
-               mc->mc_dbi, root, mc->mc_pg[0]->mp_flags));
+       DPRINTF(("db %d root page %"Z"u has flags 0x%X",
+               DDBI(mc), root, mc->mc_pg[0]->mp_flags));
 
        if (flags & MDB_PS_MODIFY) {
                if ((rc = mdb_page_touch(mc)))
@@ -4878,7 +4903,7 @@ mdb_get(MDB_txn *txn, MDB_dbi dbi,
        if (txn->mt_flags & MDB_TXN_ERROR)
                return MDB_BAD_TXN;
 
-       if (key->mv_size == 0 || key->mv_size > MDB_MAXKEYSIZE) {
+       if (key->mv_size > MDB_MAXKEYSIZE) {
                return MDB_BAD_VALSIZE;
        }
 
@@ -4930,8 +4955,11 @@ mdb_cursor_sibling(MDB_cursor *mc, int move_right)
        assert(IS_BRANCH(mc->mc_pg[mc->mc_top]));
 
        indx = NODEPTR(mc->mc_pg[mc->mc_top], mc->mc_ki[mc->mc_top]);
-       if ((rc = mdb_page_get(mc->mc_txn, NODEPGNO(indx), &mp, NULL) != 0))
+       if ((rc = mdb_page_get(mc->mc_txn, NODEPGNO(indx), &mp, NULL)) != 0) {
+               /* mc will be inconsistent if caller does mc_snum++ as above */
+               mc->mc_flags &= ~(C_INITIALIZED|C_EOF);
                return rc;
+       }
 
        mdb_cursor_push(mc, mp);
        if (!move_right)
@@ -5107,7 +5135,8 @@ mdb_cursor_set(MDB_cursor *mc, MDB_val *key, MDB_val *data,
 
        assert(mc);
        assert(key);
-       assert(key->mv_size > 0);
+       if (key->mv_size == 0)
+               return MDB_BAD_VALSIZE;
 
        if (mc->mc_xcursor)
                mc->mc_xcursor->mx_cursor.mc_flags &= ~(C_INITIALIZED|C_EOF);
@@ -5391,8 +5420,9 @@ mdb_cursor_get(MDB_cursor *mc, MDB_val *key, MDB_val *data,
                        rc = EINVAL;
                } else {
                        MDB_page *mp = mc->mc_pg[mc->mc_top];
-                       if (!NUMKEYS(mp)) {
-                               mc->mc_ki[mc->mc_top] = 0;
+                       int nkeys = NUMKEYS(mp);
+                       if (!nkeys || mc->mc_ki[mc->mc_top] >= nkeys) {
+                               mc->mc_ki[mc->mc_top] = nkeys;
                                rc = MDB_NOTFOUND;
                                break;
                        }
@@ -5431,7 +5461,7 @@ mdb_cursor_get(MDB_cursor *mc, MDB_val *key, MDB_val *data,
        case MDB_SET_RANGE:
                if (key == NULL) {
                        rc = EINVAL;
-               } else if (key->mv_size == 0 || key->mv_size > MDB_MAXKEYSIZE) {
+               } else if (key->mv_size > MDB_MAXKEYSIZE) {
                        rc = MDB_BAD_VALSIZE;
                } else if (op == MDB_SET_RANGE)
                        rc = mdb_cursor_set(mc, key, data, op, NULL);
@@ -5537,14 +5567,14 @@ fetchm:
        return rc;
 }
 
-/** Touch all the pages in the cursor stack.
+/** Touch all the pages in the cursor stack. Set mc_top.
  *     Makes sure all the pages are writable, before attempting a write operation.
  * @param[in] mc The cursor to operate on.
  */
 static int
 mdb_cursor_touch(MDB_cursor *mc)
 {
-       int rc;
+       int rc = MDB_SUCCESS;
 
        if (mc->mc_dbi > MAIN_DBI && !(*mc->mc_dbflag & DB_DIRTY)) {
                MDB_cursor mc2;
@@ -5555,13 +5585,14 @@ mdb_cursor_touch(MDB_cursor *mc)
                         return rc;
                *mc->mc_dbflag |= DB_DIRTY;
        }
-       for (mc->mc_top = 0; mc->mc_top < mc->mc_snum; mc->mc_top++) {
-               rc = mdb_page_touch(mc);
-               if (rc)
-                       return rc;
+       mc->mc_top = 0;
+       if (mc->mc_snum) {
+               do {
+                       rc = mdb_page_touch(mc);
+               } while (!rc && ++(mc->mc_top) < mc->mc_snum);
+               mc->mc_top = mc->mc_snum-1;
        }
-       mc->mc_top = mc->mc_snum-1;
-       return MDB_SUCCESS;
+       return rc;
 }
 
 /** Do not spill pages to disk if txn is getting full, may fail instead */
@@ -5612,8 +5643,8 @@ mdb_cursor_put(MDB_cursor *mc, MDB_val *key, MDB_val *data,
                return MDB_BAD_VALSIZE;
 #endif
 
-       DPRINTF(("==> put db %u key [%s], size %"Z"u, data size %"Z"u",
-               mc->mc_dbi, DKEY(key), key ? key->mv_size:0, data->mv_size));
+       DPRINTF(("==> put db %d key [%s], size %"Z"u, data size %"Z"u",
+               DDBI(mc), DKEY(key), key ? key->mv_size : 0, data->mv_size));
 
        dkey.mv_size = 0;
 
@@ -5624,6 +5655,7 @@ mdb_cursor_put(MDB_cursor *mc, MDB_val *key, MDB_val *data,
        } else if (mc->mc_db->md_root == P_INVALID) {
                /* new database, cursor has nothing to point to */
                mc->mc_snum = 0;
+               mc->mc_top = 0;
                mc->mc_flags &= ~C_INITIALIZED;
                rc = MDB_NO_ROOT;
        } else {
@@ -5942,9 +5974,6 @@ new_sub:
                        unsigned i = mc->mc_top;
                        MDB_page *mp = mc->mc_pg[i];
 
-                       if (mc->mc_flags & C_SUB)
-                               dbi--;
-
                        for (m2 = mc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                                if (mc->mc_flags & C_SUB)
                                        m3 = &m2->mc_xcursor->mx_cursor;
@@ -6041,6 +6070,7 @@ int
 mdb_cursor_del(MDB_cursor *mc, unsigned int flags)
 {
        MDB_node        *leaf;
+       MDB_page        *mp;
        int rc;
 
        if (mc->mc_txn->mt_flags & (MDB_TXN_RDONLY|MDB_TXN_ERROR))
@@ -6049,17 +6079,20 @@ mdb_cursor_del(MDB_cursor *mc, unsigned int flags)
        if (!(mc->mc_flags & C_INITIALIZED))
                return EINVAL;
 
+       if (mc->mc_ki[mc->mc_top] >= NUMKEYS(mc->mc_pg[mc->mc_top]))
+               return MDB_NOTFOUND;
+
        if (!(flags & MDB_NOSPILL) && (rc = mdb_page_spill(mc, NULL, NULL)))
                return rc;
-       flags &= ~MDB_NOSPILL; /* TODO: Or change (flags != MDB_NODUPDATA) to ~(flags & MDB_NODUPDATA), not looking at the logic of that code just now */
 
        rc = mdb_cursor_touch(mc);
        if (rc)
                return rc;
 
-       leaf = NODEPTR(mc->mc_pg[mc->mc_top], mc->mc_ki[mc->mc_top]);
+       mp = mc->mc_pg[mc->mc_top];
+       leaf = NODEPTR(mp, mc->mc_ki[mc->mc_top]);
 
-       if (!IS_LEAF2(mc->mc_pg[mc->mc_top]) && F_ISSET(leaf->mn_flags, F_DUPDATA)) {
+       if (!IS_LEAF2(mp) && F_ISSET(leaf->mn_flags, F_DUPDATA)) {
                if (!(flags & MDB_NODUPDATA)) {
                        if (!F_ISSET(leaf->mn_flags, F_SUBDATA)) {
                                mc->mc_xcursor->mx_cursor.mc_pg[0] = NODEDATA(leaf);
@@ -6074,13 +6107,13 @@ mdb_cursor_del(MDB_cursor *mc, unsigned int flags)
                                } else {
                                        MDB_cursor *m2;
                                        /* shrink fake page */
-                                       mdb_node_shrink(mc->mc_pg[mc->mc_top], mc->mc_ki[mc->mc_top]);
-                                       leaf = NODEPTR(mc->mc_pg[mc->mc_top], mc->mc_ki[mc->mc_top]);
+                                       mdb_node_shrink(mp, mc->mc_ki[mc->mc_top]);
+                                       leaf = NODEPTR(mp, mc->mc_ki[mc->mc_top]);
                                        mc->mc_xcursor->mx_cursor.mc_pg[0] = NODEDATA(leaf);
                                        /* fix other sub-DB cursors pointed at this fake page */
                                        for (m2 = mc->mc_txn->mt_cursors[mc->mc_dbi]; m2; m2=m2->mc_next) {
                                                if (m2 == mc || m2->mc_snum < mc->mc_snum) continue;
-                                               if (m2->mc_pg[mc->mc_top] == mc->mc_pg[mc->mc_top] &&
+                                               if (m2->mc_pg[mc->mc_top] == mp &&
                                                        m2->mc_ki[mc->mc_top] == mc->mc_ki[mc->mc_top])
                                                        m2->mc_xcursor->mx_cursor.mc_pg[0] = NODEDATA(leaf);
                                        }
@@ -6212,6 +6245,7 @@ mdb_node_add(MDB_cursor *mc, indx_t indx,
 {
        unsigned int     i;
        size_t           node_size = NODESIZE;
+       ssize_t          room;
        indx_t           ofs;
        MDB_node        *node;
        MDB_page        *mp = mc->mc_pg[mc->mc_top];
@@ -6242,9 +6276,9 @@ mdb_node_add(MDB_cursor *mc, indx_t indx,
                return MDB_SUCCESS;
        }
 
+       room = (ssize_t)SIZELEFT(mp) - (ssize_t)sizeof(indx_t);
        if (key != NULL)
                node_size += key->mv_size;
-
        if (IS_LEAF(mp)) {
                assert(data);
                if (F_ISSET(flags, F_BIGDATA)) {
@@ -6256,26 +6290,23 @@ mdb_node_add(MDB_cursor *mc, indx_t indx,
                        /* Put data on overflow page. */
                        DPRINTF(("data size is %"Z"u, node would be %"Z"u, put data on overflow page",
                            data->mv_size, node_size+data->mv_size));
-                       node_size += sizeof(pgno_t);
+                       node_size += sizeof(pgno_t) + (node_size & 1);
+                       if ((ssize_t)node_size > room)
+                               goto full;
                        if ((rc = mdb_page_new(mc, P_OVERFLOW, ovpages, &ofp)))
                                return rc;
                        DPRINTF(("allocated overflow page %"Z"u", ofp->mp_pgno));
                        flags |= F_BIGDATA;
+                       goto update;
                } else {
                        node_size += data->mv_size;
                }
        }
        node_size += node_size & 1;
+       if ((ssize_t)node_size > room)
+               goto full;
 
-       if (node_size + sizeof(indx_t) > SIZELEFT(mp)) {
-               DPRINTF(("not enough room in page %"Z"u, got %u ptrs",
-                   mp->mp_pgno, NUMKEYS(mp)));
-               DPRINTF(("upper - lower = %u - %u = %u", mp->mp_upper, mp->mp_lower,
-                   mp->mp_upper - mp->mp_lower));
-               DPRINTF(("node size = %"Z"u", node_size));
-               return MDB_PAGE_FULL;
-       }
-
+update:
        /* Move higher pointers up one slot. */
        for (i = NUMKEYS(mp); i > indx; i--)
                mp->mp_ptrs[i] = mp->mp_ptrs[i - 1];
@@ -6321,6 +6352,13 @@ mdb_node_add(MDB_cursor *mc, indx_t indx,
        }
 
        return MDB_SUCCESS;
+
+full:
+       DPRINTF(("not enough room in page %"Z"u, got %u ptrs",
+               mp->mp_pgno, NUMKEYS(mp)));
+       DPRINTF(("upper-lower = %u - %u = %"Z"d", mp->mp_upper,mp->mp_lower,room));
+       DPRINTF(("node size = %"Z"u", node_size));
+       return MDB_PAGE_FULL;
 }
 
 /** Delete the specified node from a page.
@@ -6455,11 +6493,13 @@ mdb_xcursor_init0(MDB_cursor *mc)
        mx->mx_cursor.mc_txn = mc->mc_txn;
        mx->mx_cursor.mc_db = &mx->mx_db;
        mx->mx_cursor.mc_dbx = &mx->mx_dbx;
-       mx->mx_cursor.mc_dbi = mc->mc_dbi+1;
+       mx->mx_cursor.mc_dbi = mc->mc_dbi;
        mx->mx_cursor.mc_dbflag = &mx->mx_dbflag;
        mx->mx_cursor.mc_snum = 0;
        mx->mx_cursor.mc_top = 0;
        mx->mx_cursor.mc_flags = C_SUB;
+       mx->mx_dbx.md_name.mv_size = 0;
+       mx->mx_dbx.md_name.mv_data = NULL;
        mx->mx_dbx.md_cmp = mc->mc_dbx->md_dcmp;
        mx->mx_dbx.md_dcmp = NULL;
        mx->mx_dbx.md_rel = mc->mc_dbx->md_rel;
@@ -6480,6 +6520,7 @@ mdb_xcursor_init1(MDB_cursor *mc, MDB_node *node)
                memcpy(&mx->mx_db, NODEDATA(node), sizeof(MDB_db));
                mx->mx_cursor.mc_pg[0] = 0;
                mx->mx_cursor.mc_snum = 0;
+               mx->mx_cursor.mc_top = 0;
                mx->mx_cursor.mc_flags = C_SUB;
        } else {
                MDB_page *fp = NODEDATA(node);
@@ -6492,8 +6533,8 @@ mdb_xcursor_init1(MDB_cursor *mc, MDB_node *node)
                mx->mx_db.md_entries = NUMKEYS(fp);
                COPY_PGNO(mx->mx_db.md_root, fp->mp_pgno);
                mx->mx_cursor.mc_snum = 1;
-               mx->mx_cursor.mc_flags = C_INITIALIZED|C_SUB;
                mx->mx_cursor.mc_top = 0;
+               mx->mx_cursor.mc_flags = C_INITIALIZED|C_SUB;
                mx->mx_cursor.mc_pg[0] = fp;
                mx->mx_cursor.mc_ki[0] = 0;
                if (mc->mc_db->md_flags & MDB_DUPFIXED) {
@@ -6503,12 +6544,9 @@ mdb_xcursor_init1(MDB_cursor *mc, MDB_node *node)
                                mx->mx_db.md_flags |= MDB_INTEGERKEY;
                }
        }
-       DPRINTF(("Sub-db %u for db %u root page %"Z"u", mx->mx_cursor.mc_dbi, mc->mc_dbi,
+       DPRINTF(("Sub-db -%u root page %"Z"u", mx->mx_cursor.mc_dbi,
                mx->mx_db.md_root));
-       mx->mx_dbflag = DB_VALID | (F_ISSET(mc->mc_pg[mc->mc_top]->mp_flags, P_DIRTY) ?
-               DB_DIRTY : 0);
-       mx->mx_dbx.md_name.mv_data = NODEKEY(node);
-       mx->mx_dbx.md_name.mv_size = node->mn_ksize;
+       mx->mx_dbflag = DB_VALID|DB_DIRTY; /* DB_DIRTY guides mdb_cursor_touch */
 #if UINT_MAX < SIZE_MAX
        if (mx->mx_dbx.md_cmp == mdb_cmp_int && mx->mx_db.md_pad == sizeof(size_t))
 #ifdef MISALIGNED_OK
@@ -6824,9 +6862,6 @@ mdb_node_move(MDB_cursor *csrc, MDB_cursor *cdst)
                MDB_dbi dbi = csrc->mc_dbi;
                MDB_page *mp = csrc->mc_pg[csrc->mc_top];
 
-               if (csrc->mc_flags & C_SUB)
-                       dbi--;
-
                for (m2 = csrc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                        if (csrc->mc_flags & C_SUB)
                                m3 = &m2->mc_xcursor->mx_cursor;
@@ -7001,9 +7036,6 @@ mdb_page_merge(MDB_cursor *csrc, MDB_cursor *cdst)
                MDB_dbi dbi = csrc->mc_dbi;
                MDB_page *mp = cdst->mc_pg[cdst->mc_top];
 
-               if (csrc->mc_flags & C_SUB)
-                       dbi--;
-
                for (m2 = csrc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                        if (csrc->mc_flags & C_SUB)
                                m3 = &m2->mc_xcursor->mx_cursor;
@@ -7098,13 +7130,11 @@ mdb_rebalance(MDB_cursor *mc)
                        /* Adjust cursors pointing to mp */
                        mc->mc_snum = 0;
                        mc->mc_top = 0;
+                       mc->mc_flags &= ~C_INITIALIZED;
                        {
                                MDB_cursor *m2, *m3;
                                MDB_dbi dbi = mc->mc_dbi;
 
-                               if (mc->mc_flags & C_SUB)
-                                       dbi--;
-
                                for (m2 = mc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                                        if (mc->mc_flags & C_SUB)
                                                m3 = &m2->mc_xcursor->mx_cursor;
@@ -7114,6 +7144,7 @@ mdb_rebalance(MDB_cursor *mc)
                                        if (m3->mc_pg[0] == mp) {
                                                m3->mc_snum = 0;
                                                m3->mc_top = 0;
+                                               m3->mc_flags &= ~C_INITIALIZED;
                                        }
                                }
                        }
@@ -7134,9 +7165,6 @@ mdb_rebalance(MDB_cursor *mc)
                                MDB_cursor *m2, *m3;
                                MDB_dbi dbi = mc->mc_dbi;
 
-                               if (mc->mc_flags & C_SUB)
-                                       dbi--;
-
                                for (m2 = mc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                                        if (mc->mc_flags & C_SUB)
                                                m3 = &m2->mc_xcursor->mx_cursor;
@@ -7144,10 +7172,13 @@ mdb_rebalance(MDB_cursor *mc)
                                                m3 = m2;
                                        if (m3 == mc || m3->mc_snum < mc->mc_snum) continue;
                                        if (m3->mc_pg[0] == mp) {
-                                               m3->mc_pg[0] = mc->mc_pg[0];
-                                               m3->mc_snum = 1;
-                                               m3->mc_top = 0;
-                                               m3->mc_ki[0] = m3->mc_ki[1];
+                                               int i;
+                                               m3->mc_snum--;
+                                               m3->mc_top--;
+                                               for (i=0; i<m3->mc_snum; i++) {
+                                                       m3->mc_pg[i] = m3->mc_pg[i+1];
+                                                       m3->mc_ki[i] = m3->mc_ki[i+1];
+                                               }
                                        }
                                }
                        }
@@ -7301,7 +7332,7 @@ mdb_del(MDB_txn *txn, MDB_dbi dbi,
        if (txn->mt_flags & (MDB_TXN_RDONLY|MDB_TXN_ERROR))
                return (txn->mt_flags & MDB_TXN_RDONLY) ? EACCES : MDB_BAD_TXN;
 
-       if (key->mv_size == 0 || key->mv_size > MDB_MAXKEYSIZE) {
+       if (key->mv_size > MDB_MAXKEYSIZE) {
                return MDB_BAD_VALSIZE;
        }
 
@@ -7354,24 +7385,26 @@ mdb_page_split(MDB_cursor *mc, MDB_val *newkey, MDB_val *newdata, pgno_t newpgno
        unsigned int nflags)
 {
        unsigned int flags;
-       int              rc = MDB_SUCCESS, ins_new = 0, new_root = 0, newpos = 1, did_split = 0;
+       int              rc = MDB_SUCCESS, new_root = 0, did_split = 0;
        indx_t           newindx;
        pgno_t           pgno = 0;
-       unsigned int     i, j, split_indx, nkeys, pmax;
+       int      i, j, split_indx, nkeys, pmax;
+       MDB_env         *env = mc->mc_txn->mt_env;
        MDB_node        *node;
        MDB_val  sepkey, rkey, xdata, *rdata = &xdata;
-       MDB_page        *copy;
+       MDB_page        *copy = NULL;
        MDB_page        *mp, *rp, *pp;
-       unsigned int ptop;
+       int ptop;
        MDB_cursor      mn;
        DKBUF;
 
        mp = mc->mc_pg[mc->mc_top];
        newindx = mc->mc_ki[mc->mc_top];
+       nkeys = NUMKEYS(mp);
 
-       DPRINTF(("-----> splitting %s page %"Z"u and adding [%s] at index %i",
+       DPRINTF(("-----> splitting %s page %"Z"u and adding [%s] at index %i/%i",
            IS_LEAF(mp) ? "leaf" : "branch", mp->mp_pgno,
-           DKEY(newkey), mc->mc_ki[mc->mc_top]));
+           DKEY(newkey), mc->mc_ki[mc->mc_top], nkeys));
 
        /* Create a right sibling. */
        if ((rc = mdb_page_new(mc, mp->mp_flags, 1, &rp)))
@@ -7418,141 +7451,146 @@ mdb_page_split(MDB_cursor *mc, MDB_val *newkey, MDB_val *newdata, pgno_t newpgno
                sepkey = *newkey;
                split_indx = newindx;
                nkeys = 0;
-               goto newsep;
-       }
+       } else {
 
-       nkeys = NUMKEYS(mp);
-       split_indx = nkeys / 2;
-       if (newindx < split_indx)
-               newpos = 0;
-
-       if (IS_LEAF2(rp)) {
-               char *split, *ins;
-               int x;
-               unsigned int lsize, rsize, ksize;
-               /* Move half of the keys to the right sibling */
-               copy = NULL;
-               x = mc->mc_ki[mc->mc_top] - split_indx;
-               ksize = mc->mc_db->md_pad;
-               split = LEAF2KEY(mp, split_indx, ksize);
-               rsize = (nkeys - split_indx) * ksize;
-               lsize = (nkeys - split_indx) * sizeof(indx_t);
-               mp->mp_lower -= lsize;
-               rp->mp_lower += lsize;
-               mp->mp_upper += rsize - lsize;
-               rp->mp_upper -= rsize - lsize;
-               sepkey.mv_size = ksize;
-               if (newindx == split_indx) {
-                       sepkey.mv_data = newkey->mv_data;
-               } else {
-                       sepkey.mv_data = split;
-               }
-               if (x<0) {
-                       ins = LEAF2KEY(mp, mc->mc_ki[mc->mc_top], ksize);
-                       memcpy(rp->mp_ptrs, split, rsize);
-                       sepkey.mv_data = rp->mp_ptrs;
-                       memmove(ins+ksize, ins, (split_indx - mc->mc_ki[mc->mc_top]) * ksize);
-                       memcpy(ins, newkey->mv_data, ksize);
-                       mp->mp_lower += sizeof(indx_t);
-                       mp->mp_upper -= ksize - sizeof(indx_t);
+               split_indx = (nkeys+1) / 2;
+
+               if (IS_LEAF2(rp)) {
+                       char *split, *ins;
+                       int x;
+                       unsigned int lsize, rsize, ksize;
+                       /* Move half of the keys to the right sibling */
+                       copy = NULL;
+                       x = mc->mc_ki[mc->mc_top] - split_indx;
+                       ksize = mc->mc_db->md_pad;
+                       split = LEAF2KEY(mp, split_indx, ksize);
+                       rsize = (nkeys - split_indx) * ksize;
+                       lsize = (nkeys - split_indx) * sizeof(indx_t);
+                       mp->mp_lower -= lsize;
+                       rp->mp_lower += lsize;
+                       mp->mp_upper += rsize - lsize;
+                       rp->mp_upper -= rsize - lsize;
+                       sepkey.mv_size = ksize;
+                       if (newindx == split_indx) {
+                               sepkey.mv_data = newkey->mv_data;
+                       } else {
+                               sepkey.mv_data = split;
+                       }
+                       if (x<0) {
+                               ins = LEAF2KEY(mp, mc->mc_ki[mc->mc_top], ksize);
+                               memcpy(rp->mp_ptrs, split, rsize);
+                               sepkey.mv_data = rp->mp_ptrs;
+                               memmove(ins+ksize, ins, (split_indx - mc->mc_ki[mc->mc_top]) * ksize);
+                               memcpy(ins, newkey->mv_data, ksize);
+                               mp->mp_lower += sizeof(indx_t);
+                               mp->mp_upper -= ksize - sizeof(indx_t);
+                       } else {
+                               if (x)
+                                       memcpy(rp->mp_ptrs, split, x * ksize);
+                               ins = LEAF2KEY(rp, x, ksize);
+                               memcpy(ins, newkey->mv_data, ksize);
+                               memcpy(ins+ksize, split + x * ksize, rsize - x * ksize);
+                               rp->mp_lower += sizeof(indx_t);
+                               rp->mp_upper -= ksize - sizeof(indx_t);
+                               mc->mc_ki[mc->mc_top] = x;
+                               mc->mc_pg[mc->mc_top] = rp;
+                       }
                } else {
-                       if (x)
-                               memcpy(rp->mp_ptrs, split, x * ksize);
-                       ins = LEAF2KEY(rp, x, ksize);
-                       memcpy(ins, newkey->mv_data, ksize);
-                       memcpy(ins+ksize, split + x * ksize, rsize - x * ksize);
-                       rp->mp_lower += sizeof(indx_t);
-                       rp->mp_upper -= ksize - sizeof(indx_t);
-                       mc->mc_ki[mc->mc_top] = x;
-                       mc->mc_pg[mc->mc_top] = rp;
-               }
-               goto newsep;
-       }
+                       int psize, nsize, k;
+                       /* Maximum free space in an empty page */
+                       pmax = env->me_psize - PAGEHDRSZ;
+                       if (IS_LEAF(mp))
+                               nsize = mdb_leaf_size(env, newkey, newdata);
+                       else
+                               nsize = mdb_branch_size(env, newkey);
+                       nsize += nsize & 1;
 
-       /* For leaf pages, check the split point based on what
-        * fits where, since otherwise mdb_node_add can fail.
-        *
-        * This check is only needed when the data items are
-        * relatively large, such that being off by one will
-        * make the difference between success or failure.
-        *
-        * It's also relevant if a page happens to be laid out
-        * such that one half of its nodes are all "small" and
-        * the other half of its nodes are "large." If the new
-        * item is also "large" and falls on the half with
-        * "large" nodes, it also may not fit.
-        */
-       if (IS_LEAF(mp)) {
-               unsigned int psize, nsize;
-               /* Maximum free space in an empty page */
-               pmax = mc->mc_txn->mt_env->me_psize - PAGEHDRSZ;
-               nsize = mdb_leaf_size(mc->mc_txn->mt_env, newkey, newdata);
-               if ((nkeys < 20) || (nsize > pmax/16)) {
-                       if (newindx <= split_indx) {
-                               psize = nsize;
-                               newpos = 0;
-                               for (i=0; i<split_indx; i++) {
-                                       node = NODEPTR(mp, i);
-                                       psize += NODESIZE + NODEKSZ(node) + sizeof(indx_t);
-                                       if (F_ISSET(node->mn_flags, F_BIGDATA))
-                                               psize += sizeof(pgno_t);
-                                       else
-                                               psize += NODEDSZ(node);
-                                       psize += psize & 1;
-                                       if (psize > pmax) {
-                                               if (i <= newindx) {
-                                                       split_indx = newindx;
-                                                       if (i < newindx)
-                                                               newpos = 1;
+                       /* grab a page to hold a temporary copy */
+                       copy = mdb_page_malloc(mc->mc_txn, 1);
+                       if (copy == NULL)
+                               return ENOMEM;
+                       copy->mp_pgno  = mp->mp_pgno;
+                       copy->mp_flags = mp->mp_flags;
+                       copy->mp_lower = PAGEHDRSZ;
+                       copy->mp_upper = env->me_psize;
+
+                       /* prepare to insert */
+                       for (i=0, j=0; i<nkeys; i++) {
+                               if (i == newindx) {
+                                       copy->mp_ptrs[j++] = 0;
+                               }
+                               copy->mp_ptrs[j++] = mp->mp_ptrs[i];
+                       }
+
+                       /* When items are relatively large the split point needs
+                        * to be checked, because being off-by-one will make the
+                        * difference between success or failure in mdb_node_add.
+                        *
+                        * It's also relevant if a page happens to be laid out
+                        * such that one half of its nodes are all "small" and
+                        * the other half of its nodes are "large." If the new
+                        * item is also "large" and falls on the half with
+                        * "large" nodes, it also may not fit.
+                        *
+                        * As a final tweak, if the new item goes on the last
+                        * spot on the page (and thus, onto the new page), bias
+                        * the split so the new page is emptier than the old page.
+                        * This yields better packing during sequential inserts.
+                        */
+                       if (nkeys < 20 || nsize > pmax/16 || newindx >= nkeys) {
+                               /* Find split point */
+                               psize = 0;
+                               if (newindx <= split_indx || newindx >= nkeys) {
+                                       i = 0; j = 1;
+                                       k = newindx >= nkeys ? nkeys : split_indx+1;
+                               } else {
+                                       i = nkeys; j = -1;
+                                       k = split_indx-1;
+                               }
+                               for (; i!=k; i+=j) {
+                                       if (i == newindx) {
+                                               psize += nsize;
+                                               node = NULL;
+                                       } else {
+                                               node = (MDB_node *)((char *)mp + copy->mp_ptrs[i]);
+                                               psize += NODESIZE + NODEKSZ(node) + sizeof(indx_t);
+                                               if (IS_LEAF(mp)) {
+                                                       if (F_ISSET(node->mn_flags, F_BIGDATA))
+                                                               psize += sizeof(pgno_t);
+                                                       else
+                                                               psize += NODEDSZ(node);
                                                }
-                                               else
-                                                       split_indx = i;
-                                               break;
+                                               psize += psize & 1;
                                        }
-                               }
-                       } else {
-                               psize = nsize;
-                               for (i=nkeys-1; i>=split_indx; i--) {
-                                       node = NODEPTR(mp, i);
-                                       psize += NODESIZE + NODEKSZ(node) + sizeof(indx_t);
-                                       if (F_ISSET(node->mn_flags, F_BIGDATA))
-                                               psize += sizeof(pgno_t);
-                                       else
-                                               psize += NODEDSZ(node);
-                                       psize += psize & 1;
                                        if (psize > pmax) {
-                                               if (i >= newindx) {
-                                                       split_indx = newindx;
-                                                       newpos = 0;
-                                               } else
-                                                       split_indx = i+1;
+                                               split_indx = i + (j<0);
                                                break;
                                        }
                                }
+                               /* special case: when the new node was on the last
+                                * slot we may not have tripped the break inside the loop.
+                                * In all other cases we either hit the break condition,
+                                * or the original split_indx was already safe.
+                                */
+                               if (newindx >= nkeys && i == k)
+                                       split_indx = nkeys-1;
+                       }
+                       if (split_indx == newindx) {
+                               sepkey.mv_size = newkey->mv_size;
+                               sepkey.mv_data = newkey->mv_data;
+                       } else {
+                               node = (MDB_node *)((char *)mp + copy->mp_ptrs[split_indx]);
+                               sepkey.mv_size = node->mn_ksize;
+                               sepkey.mv_data = NODEKEY(node);
                        }
                }
        }
 
-       /* First find the separating key between the split pages.
-        * The case where newindx == split_indx is ambiguous; the
-        * new item could go to the new page or stay on the original
-        * page. If newpos == 1 it goes to the new page.
-        */
-       if (newindx == split_indx && newpos) {
-               sepkey.mv_size = newkey->mv_size;
-               sepkey.mv_data = newkey->mv_data;
-       } else {
-               node = NODEPTR(mp, split_indx);
-               sepkey.mv_size = node->mn_ksize;
-               sepkey.mv_data = NODEKEY(node);
-       }
-
-newsep:
-       DPRINTF(("separator is [%s]", DKEY(&sepkey)));
+       DPRINTF(("separator is %d [%s]", split_indx, DKEY(&sepkey)));
 
        /* Copy separator key to the parent.
         */
-       if (SIZELEFT(mn.mc_pg[ptop]) < mdb_branch_size(mc->mc_txn->mt_env, &sepkey)) {
+       if (SIZELEFT(mn.mc_pg[ptop]) < mdb_branch_size(env, &sepkey)) {
                mn.mc_snum--;
                mn.mc_top--;
                did_split = 1;
@@ -7597,117 +7635,97 @@ newsep:
                        return rc;
                for (i=0; i<mc->mc_top; i++)
                        mc->mc_ki[i] = mn.mc_ki[i];
-               goto done;
-       }
-       if (IS_LEAF2(rp)) {
-               goto done;
-       }
-
-       /* Move half of the keys to the right sibling. */
+       } else if (!IS_LEAF2(mp)) {
+               /* Move nodes */
+               mc->mc_pg[mc->mc_top] = rp;
+               i = split_indx;
+               j = 0;
+               do {
+                       if (i == newindx) {
+                               rkey.mv_data = newkey->mv_data;
+                               rkey.mv_size = newkey->mv_size;
+                               if (IS_LEAF(mp)) {
+                                       rdata = newdata;
+                               } else
+                                       pgno = newpgno;
+                               flags = nflags;
+                               /* Update index for the new key. */
+                               mc->mc_ki[mc->mc_top] = j;
+                       } else {
+                               node = (MDB_node *)((char *)mp + copy->mp_ptrs[i]);
+                               rkey.mv_data = NODEKEY(node);
+                               rkey.mv_size = node->mn_ksize;
+                               if (IS_LEAF(mp)) {
+                                       xdata.mv_data = NODEDATA(node);
+                                       xdata.mv_size = NODEDSZ(node);
+                                       rdata = &xdata;
+                               } else
+                                       pgno = NODEPGNO(node);
+                               flags = node->mn_flags;
+                       }
 
-       /* grab a page to hold a temporary copy */
-       copy = mdb_page_malloc(mc->mc_txn, 1);
-       if (copy == NULL)
-               return ENOMEM;
+                       if (!IS_LEAF(mp) && j == 0) {
+                               /* First branch index doesn't need key data. */
+                               rkey.mv_size = 0;
+                       }
 
-       copy->mp_pgno  = mp->mp_pgno;
-       copy->mp_flags = mp->mp_flags;
-       copy->mp_lower = PAGEHDRSZ;
-       copy->mp_upper = mc->mc_txn->mt_env->me_psize;
-       mc->mc_pg[mc->mc_top] = copy;
-       for (i = j = 0; i <= nkeys; j++) {
-               if (i == split_indx) {
-               /* Insert in right sibling. */
-               /* Reset insert index for right sibling. */
-                       if (i != newindx || (newpos ^ ins_new)) {
+                       rc = mdb_node_add(mc, j, &rkey, rdata, pgno, flags);
+                       if (rc) {
+                               /* return tmp page to freelist */
+                               mdb_page_free(env, copy);
+                               return rc;
+                       }
+                       if (i == nkeys) {
+                               i = 0;
                                j = 0;
-                               mc->mc_pg[mc->mc_top] = rp;
+                               mc->mc_pg[mc->mc_top] = copy;
+                       } else {
+                               i++;
+                               j++;
+                       }
+               } while (i != split_indx);
+
+               nkeys = NUMKEYS(copy);
+               for (i=0; i<nkeys; i++)
+                       mp->mp_ptrs[i] = copy->mp_ptrs[i];
+               mp->mp_lower = copy->mp_lower;
+               mp->mp_upper = copy->mp_upper;
+               memcpy(NODEPTR(mp, nkeys-1), NODEPTR(copy, nkeys-1),
+                       env->me_psize - copy->mp_upper);
+
+               /* reset back to original page */
+               if (newindx < split_indx) {
+                       mc->mc_pg[mc->mc_top] = mp;
+                       if (nflags & MDB_RESERVE) {
+                               node = NODEPTR(mp, mc->mc_ki[mc->mc_top]);
+                               if (!(node->mn_flags & F_BIGDATA))
+                                       newdata->mv_data = NODEDATA(node);
                        }
-               }
-
-               if (i == newindx && !ins_new) {
-                       /* Insert the original entry that caused the split. */
-                       rkey.mv_data = newkey->mv_data;
-                       rkey.mv_size = newkey->mv_size;
-                       if (IS_LEAF(mp)) {
-                               rdata = newdata;
-                       } else
-                               pgno = newpgno;
-                       flags = nflags;
-
-                       ins_new = 1;
-
-                       /* Update index for the new key. */
-                       mc->mc_ki[mc->mc_top] = j;
-               } else if (i == nkeys) {
-                       break;
                } else {
-                       node = NODEPTR(mp, i);
-                       rkey.mv_data = NODEKEY(node);
-                       rkey.mv_size = node->mn_ksize;
-                       if (IS_LEAF(mp)) {
-                               xdata.mv_data = NODEDATA(node);
-                               xdata.mv_size = NODEDSZ(node);
-                               rdata = &xdata;
-                       } else
-                               pgno = NODEPGNO(node);
-                       flags = node->mn_flags;
-
-                       i++;
-               }
-
-               if (!IS_LEAF(mp) && j == 0) {
-                       /* First branch index doesn't need key data. */
-                       rkey.mv_size = 0;
-               }
-
-               rc = mdb_node_add(mc, j, &rkey, rdata, pgno, flags);
-               if (rc) break;
-       }
-
-       nkeys = NUMKEYS(copy);
-       for (i=0; i<nkeys; i++)
-               mp->mp_ptrs[i] = copy->mp_ptrs[i];
-       mp->mp_lower = copy->mp_lower;
-       mp->mp_upper = copy->mp_upper;
-       memcpy(NODEPTR(mp, nkeys-1), NODEPTR(copy, nkeys-1),
-               mc->mc_txn->mt_env->me_psize - copy->mp_upper);
-
-       /* reset back to original page */
-       if (newindx < split_indx || (!newpos && newindx == split_indx)) {
-               mc->mc_pg[mc->mc_top] = mp;
-               if (nflags & MDB_RESERVE) {
-                       node = NODEPTR(mp, mc->mc_ki[mc->mc_top]);
-                       if (!(node->mn_flags & F_BIGDATA))
-                               newdata->mv_data = NODEDATA(node);
-               }
-       } else {
-               mc->mc_ki[ptop]++;
-               /* Make sure mc_ki is still valid.
-                */
-               if (mn.mc_pg[ptop] != mc->mc_pg[ptop] &&
-                   mc->mc_ki[ptop] >= NUMKEYS(mc->mc_pg[ptop])) {
-                       for (i=0; i<ptop; i++) {
-                               mc->mc_pg[i] = mn.mc_pg[i];
-                               mc->mc_ki[i] = mn.mc_ki[i];
+                       mc->mc_pg[mc->mc_top] = rp;
+                       mc->mc_ki[ptop]++;
+                       /* Make sure mc_ki is still valid.
+                        */
+                       if (mn.mc_pg[ptop] != mc->mc_pg[ptop] &&
+                               mc->mc_ki[ptop] >= NUMKEYS(mc->mc_pg[ptop])) {
+                               for (i=0; i<ptop; i++) {
+                                       mc->mc_pg[i] = mn.mc_pg[i];
+                                       mc->mc_ki[i] = mn.mc_ki[i];
+                               }
+                               mc->mc_pg[ptop] = mn.mc_pg[ptop];
+                               mc->mc_ki[ptop] = mn.mc_ki[ptop] - 1;
                        }
-                       mc->mc_pg[ptop] = mn.mc_pg[ptop];
-                       mc->mc_ki[ptop] = mn.mc_ki[ptop] - 1;
                }
+               /* return tmp page to freelist */
+               mdb_page_free(env, copy);
        }
 
-       /* return tmp page to freelist */
-       mdb_page_free(mc->mc_txn->mt_env, copy);
-done:
        {
                /* Adjust other cursors pointing to mp */
                MDB_cursor *m2, *m3;
                MDB_dbi dbi = mc->mc_dbi;
                int fixup = NUMKEYS(mp);
 
-               if (mc->mc_flags & C_SUB)
-                       dbi--;
-
                for (m2 = mc->mc_txn->mt_cursors[dbi]; m2; m2=m2->mc_next) {
                        if (mc->mc_flags & C_SUB)
                                m3 = &m2->mc_xcursor->mx_cursor;
@@ -7749,6 +7767,7 @@ done:
                        }
                }
        }
+       DPRINTF(("mp left: %d, rp left: %d", SIZELEFT(mp), SIZELEFT(rp)));
        return rc;
 }
 
@@ -7765,13 +7784,6 @@ mdb_put(MDB_txn *txn, MDB_dbi dbi,
        if (txn == NULL || !dbi || dbi >= txn->mt_numdbs || !(txn->mt_dbflags[dbi] & DB_VALID))
                return EINVAL;
 
-       if (txn->mt_flags & (MDB_TXN_RDONLY|MDB_TXN_ERROR))
-               return (txn->mt_flags & MDB_TXN_RDONLY) ? EACCES : MDB_BAD_TXN;
-
-       if (key->mv_size == 0 || key->mv_size > MDB_MAXKEYSIZE) {
-               return MDB_BAD_VALSIZE;
-       }
-
        if ((flags & (MDB_NOOVERWRITE|MDB_NODUPDATA|MDB_RESERVE|MDB_APPEND|MDB_APPENDDUP)) != flags)
                return EINVAL;
 
@@ -7811,6 +7823,16 @@ mdb_env_get_path(MDB_env *env, const char **arg)
        return MDB_SUCCESS;
 }
 
+int
+mdb_env_get_fd(MDB_env *env, mdb_filehandle_t *arg)
+{
+       if (!env || !arg)
+               return EINVAL;
+
+       *arg = env->me_fd;
+       return MDB_SUCCESS;
+}
+
 /** Common code for #mdb_stat() and #mdb_env_stat().
  * @param[in] env the environment to operate in.
  * @param[in] db the #MDB_db record containing the stats to return.
@@ -8261,7 +8283,7 @@ static int mdb_pid_insert(pid_t *ids, pid_t pid)
                        return -1;
                }
        }
-       
+
        if( val > 0 ) {
                ++cursor;
        }