removed a bug introduced with non blocking VM

[redis.git] / redis.c
diff --git a/redis.c b/redis.c

index 9f1eb7a2507d0ba4d3d466ce7e67d4880bd038fa..f489072e26c9872238cb3846f21a9ff5fea9f11e 100644 (file)
--- a/redis.c
+++ b/redis.c
@@ -27,7 +27,7 @@
   * POSSIBILITY OF SUCH DAMAGE.
   */
  
   * POSSIBILITY OF SUCH DAMAGE.
   */
  
-#define REDIS_VERSION "1.3.0"
+#define REDIS_VERSION "1.3.2"
  
  #include "fmacros.h"
  #include "config.h"
  
  #include "fmacros.h"
  #include "config.h"
@@ -59,6 +59,7 @@
  #include <sys/uio.h>
  #include <limits.h>
  #include <math.h>
  #include <sys/uio.h>
  #include <limits.h>
  #include <math.h>
+#include <pthread.h>
  
  #if defined(__sun)
  #include "solarisfixes.h"
  
  #if defined(__sun)
  #include "solarisfixes.h"
@@ -152,6 +153,18 @@
  #define REDIS_RDB_ENC_INT32 2       /* 32 bit signed integer */
  #define REDIS_RDB_ENC_LZF 3         /* string compressed with FASTLZ */
  
  #define REDIS_RDB_ENC_INT32 2       /* 32 bit signed integer */
  #define REDIS_RDB_ENC_LZF 3         /* string compressed with FASTLZ */
  
+/* Virtual memory object->where field. */
+#define REDIS_VM_MEMORY 0       /* The object is on memory */
+#define REDIS_VM_SWAPPED 1      /* The object is on disk */
+#define REDIS_VM_SWAPPING 2     /* Redis is swapping this object on disk */
+#define REDIS_VM_LOADING 3      /* Redis is loading this object from disk */
+
+/* Virtual memory static configuration stuff.
+ * Check vmFindContiguousPages() to know more about this magic numbers. */
+#define REDIS_VM_MAX_NEAR_PAGES 65536
+#define REDIS_VM_MAX_RANDOM_JUMP 4096
+#define REDIS_VM_MAX_THREADS 32
+
  /* Client flags */
  #define REDIS_CLOSE 1       /* This client connection should be closed ASAP */
  #define REDIS_SLAVE 2       /* This client is a slave server */
  /* Client flags */
  #define REDIS_CLOSE 1       /* This client connection should be closed ASAP */
  #define REDIS_SLAVE 2       /* This client is a slave server */
@@ -159,6 +172,7 @@
  #define REDIS_MONITOR 8      /* This client is a slave monitor, see MONITOR */
  #define REDIS_MULTI 16      /* This client is in a MULTI context */
  #define REDIS_BLOCKED 32    /* The client is waiting in a blocking operation */
  #define REDIS_MONITOR 8      /* This client is a slave monitor, see MONITOR */
  #define REDIS_MULTI 16      /* This client is in a MULTI context */
  #define REDIS_BLOCKED 32    /* The client is waiting in a blocking operation */
+#define REDIS_IO_WAIT 64    /* The client is waiting for Virtual Memory I/O */
  
  /* Slave replication state - slave side */
  #define REDIS_REPL_NONE 0   /* No active replication */
  
  /* Slave replication state - slave side */
  #define REDIS_REPL_NONE 0   /* No active replication */
@@ -186,8 +200,9 @@
  
  /* Log levels */
  #define REDIS_DEBUG 0
  
  /* Log levels */
  #define REDIS_DEBUG 0
-#define REDIS_NOTICE 1
-#define REDIS_WARNING 2
+#define REDIS_VERBOSE 1
+#define REDIS_NOTICE 2
+#define REDIS_WARNING 3
  
  /* Anti-warning macro... */
  #define REDIS_NOTUSED(V) ((void) V)
  
  /* Anti-warning macro... */
  #define REDIS_NOTUSED(V) ((void) V)
@@ -207,12 +222,29 @@ static void _redisAssert(char *estr);
  /*================================= Data types ============================== */
  
  /* A redis object, that is a type able to hold a string / list / set */
  /*================================= Data types ============================== */
  
  /* A redis object, that is a type able to hold a string / list / set */
+
+/* The VM object structure */
+struct redisObjectVM {
+    off_t page;         /* the page at witch the object is stored on disk */
+    off_t usedpages;    /* number of pages used on disk */
+    time_t atime;       /* Last access time */
+} vm;
+
+/* The actual Redis Object */
  typedef struct redisObject {
      void *ptr;
      unsigned char type;
      unsigned char encoding;
  typedef struct redisObject {
      void *ptr;
      unsigned char type;
      unsigned char encoding;
-    unsigned char notused[2];
+    unsigned char storage;  /* If this object is a key, where is the value?
+                             * REDIS_VM_MEMORY, REDIS_VM_SWAPPED, ... */
+    unsigned char vtype; /* If this object is a key, and value is swapped out,
+                          * this is the type of the swapped out object. */
      int refcount;
      int refcount;
+    /* VM fields, this are only allocated if VM is active, otherwise the
+     * object allocation function will just allocate
+     * sizeof(redisObjct) minus sizeof(redisObjectVM), so using
+     * Redis without VM active will not have any overhead. */
+    struct redisObjectVM vm;
  } robj;
  
  /* Macro used to initalize a Redis object allocated on the stack.
  } robj;
  
  /* Macro used to initalize a Redis object allocated on the stack.
@@ -224,6 +256,7 @@ typedef struct redisObject {
      _var.type = REDIS_STRING; \
      _var.encoding = REDIS_ENCODING_RAW; \
      _var.ptr = _ptr; \
      _var.type = REDIS_STRING; \
      _var.encoding = REDIS_ENCODING_RAW; \
      _var.ptr = _ptr; \
+    if (server.vm_enabled) _var.storage = REDIS_VM_MEMORY; \
  } while(0);
  
  typedef struct redisDb {
  } while(0);
  
  typedef struct redisDb {
@@ -273,6 +306,8 @@ typedef struct redisClient {
      int blockingkeysnum;    /* Number of blocking keys */
      time_t blockingto;      /* Blocking operation timeout. If UNIX current time
                               * is >= blockingto then the operation timed out. */
      int blockingkeysnum;    /* Number of blocking keys */
      time_t blockingto;      /* Blocking operation timeout. If UNIX current time
                               * is >= blockingto then the operation timed out. */
+    list *io_keys;          /* Keys this client is waiting to be loaded from the
+                             * swap file in order to continue. */
  } redisClient;
  
  struct saveparam {
  } redisClient;
  
  struct saveparam {
@@ -332,13 +367,48 @@ struct redisServer {
      redisClient *master;    /* client that is master for this slave */
      int replstate;
      unsigned int maxclients;
      redisClient *master;    /* client that is master for this slave */
      int replstate;
      unsigned int maxclients;
-    unsigned long maxmemory;
+    unsigned long long maxmemory;
      unsigned int blockedclients;
      /* Sort parameters - qsort_r() is only available under BSD so we
       * have to take this state global, in order to pass it to sortCompare() */
      int sort_desc;
      int sort_alpha;
      int sort_bypattern;
      unsigned int blockedclients;
      /* Sort parameters - qsort_r() is only available under BSD so we
       * have to take this state global, in order to pass it to sortCompare() */
      int sort_desc;
      int sort_alpha;
      int sort_bypattern;
+    /* Virtual memory configuration */
+    int vm_enabled;
+    off_t vm_page_size;
+    off_t vm_pages;
+    unsigned long long vm_max_memory;
+    /* Virtual memory state */
+    FILE *vm_fp;
+    int vm_fd;
+    off_t vm_next_page; /* Next probably empty page */
+    off_t vm_near_pages; /* Number of pages allocated sequentially */
+    unsigned char *vm_bitmap; /* Bitmap of free/used pages */
+    time_t unixtime;    /* Unix time sampled every second. */
+    /* Virtual memory I/O threads stuff */
+    /* An I/O thread process an element taken from the io_jobs queue and
+     * put the result of the operation in the io_done list. While the
+     * job is being processed, it's put on io_processing queue. */
+    list *io_newjobs; /* List of VM I/O jobs yet to be processed */
+    list *io_processing; /* List of VM I/O jobs being processed */
+    list *io_processed; /* List of VM I/O jobs already processed */
+    list *io_clients; /* All the clients waiting for SWAP I/O operations */
+    pthread_mutex_t io_mutex; /* lock to access io_jobs/io_done/io_thread_job */
+    int io_active_threads; /* Number of running I/O threads */
+    int vm_max_threads; /* Max number of I/O threads running at the same time */
+    /* Our main thread is blocked on the event loop, locking for sockets ready
+     * to be read or written, so when a threaded I/O operation is ready to be
+     * processed by the main thread, the I/O thread will use a unix pipe to
+     * awake the main thread. The followings are the two pipe FDs. */
+    int io_ready_pipe_read;
+    int io_ready_pipe_write;
+    /* Virtual memory stats */
+    unsigned long long vm_stats_used_pages;
+    unsigned long long vm_stats_swapped_objects;
+    unsigned long long vm_stats_swapouts;
+    unsigned long long vm_stats_swapins;
+    FILE *devnull;
  };
  
  typedef void redisCommandProc(redisClient *c);
  };
  
  typedef void redisCommandProc(redisClient *c);
@@ -404,6 +474,22 @@ struct sharedObjectsStruct {
  
  static double R_Zero, R_PosInf, R_NegInf, R_Nan;
  
  
  static double R_Zero, R_PosInf, R_NegInf, R_Nan;
  
+/* VM threaded I/O request message */
+#define REDIS_IOJOB_LOAD 0          /* Load from disk to memory */
+#define REDIS_IOJOB_PREPARE_SWAP 1  /* Compute needed pages */
+#define REDIS_IOJOB_DO_SWAP 2       /* Swap from memory to disk */
+typedef struct iojon {
+    int type;   /* Request type, REDIS_IOJOB_* */
+    redisDb *db;/* Redis database */
+    robj *key;  /* This I/O request is about swapping this key */
+    robj *val;  /* the value to swap for REDIS_IOREQ_*_SWAP, otherwise this
+                 * field is populated by the I/O thread for REDIS_IOREQ_LOAD. */
+    off_t page; /* Swap page where to read/write the object */
+    off_t pages; /* Swap pages needed to safe object. PREPARE_SWAP return val */
+    int canceled; /* True if this command was canceled by blocking side of VM */
+    pthread_t thread; /* ID of the thread processing this entry */
+} iojob;
+
  /*================================ Prototypes =============================== */
  
  static void freeStringObject(robj *o);
  /*================================ Prototypes =============================== */
  
  static void freeStringObject(robj *o);
@@ -418,6 +504,7 @@ static void addReplySds(redisClient *c, sds s);
  static void incrRefCount(robj *o);
  static int rdbSaveBackground(char *filename);
  static robj *createStringObject(char *ptr, size_t len);
  static void incrRefCount(robj *o);
  static int rdbSaveBackground(char *filename);
  static robj *createStringObject(char *ptr, size_t len);
+static robj *dupStringObject(robj *o);
  static void replicationFeedSlaves(list *slaves, struct redisCommand *cmd, int dictid, robj **argv, int argc);
  static void feedAppendOnlyFile(struct redisCommand *cmd, int dictid, robj **argv, int argc);
  static int syncWithMaster(void);
  static void replicationFeedSlaves(list *slaves, struct redisCommand *cmd, int dictid, robj **argv, int argc);
  static void feedAppendOnlyFile(struct redisCommand *cmd, int dictid, robj **argv, int argc);
  static int syncWithMaster(void);
@@ -427,6 +514,7 @@ static robj *getDecodedObject(robj *o);
  static int removeExpire(redisDb *db, robj *key);
  static int expireIfNeeded(redisDb *db, robj *key);
  static int deleteIfVolatile(redisDb *db, robj *key);
  static int removeExpire(redisDb *db, robj *key);
  static int expireIfNeeded(redisDb *db, robj *key);
  static int deleteIfVolatile(redisDb *db, robj *key);
+static int deleteIfSwapped(redisDb *db, robj *key);
  static int deleteKey(redisDb *db, robj *key);
  static time_t getExpire(redisDb *db, robj *key);
  static int setExpire(redisDb *db, robj *key, time_t when);
  static int deleteKey(redisDb *db, robj *key);
  static time_t getExpire(redisDb *db, robj *key);
  static int setExpire(redisDb *db, robj *key, time_t when);
@@ -447,6 +535,22 @@ static void freeClientMultiState(redisClient *c);
  static void queueMultiCommand(redisClient *c, struct redisCommand *cmd);
  static void unblockClient(redisClient *c);
  static int handleClientsWaitingListPush(redisClient *c, robj *key, robj *ele);
  static void queueMultiCommand(redisClient *c, struct redisCommand *cmd);
  static void unblockClient(redisClient *c);
  static int handleClientsWaitingListPush(redisClient *c, robj *key, robj *ele);
+static void vmInit(void);
+static void vmMarkPagesFree(off_t page, off_t count);
+static robj *vmLoadObject(robj *key);
+static robj *vmPreviewObject(robj *key);
+static int vmSwapOneObjectBlocking(void);
+static int vmSwapOneObjectThreaded(void);
+static int vmCanSwapOut(void);
+static void freeOneObjectFromFreelist(void);
+static void acceptHandler(aeEventLoop *el, int fd, void *privdata, int mask);
+static void vmThreadedIOCompletedJob(aeEventLoop *el, int fd, void *privdata, int mask);
+static void vmCancelThreadedIOJob(robj *o);
+static void lockThreadedIO(void);
+static void unlockThreadedIO(void);
+static int vmSwapObjectThreaded(robj *key, robj *val, redisDb *db);
+static void freeIOJob(iojob *j);
+static void queueIOJob(iojob *j);
  
  static void authCommand(redisClient *c);
  static void pingCommand(redisClient *c);
  
  static void authCommand(redisClient *c);
  static void pingCommand(redisClient *c);
@@ -796,6 +900,7 @@ static void dictRedisObjectDestructor(void *privdata, void *val)
  {
      DICT_NOTUSED(privdata);
  
  {
      DICT_NOTUSED(privdata);
  
+    if (val == NULL) return; /* Values of swapped out keys as set to NULL */
      decrRefCount(val);
  }
  
      decrRefCount(val);
  }
  
@@ -899,7 +1004,7 @@ static void closeTimedoutClients(void) {
              !(c->flags & REDIS_MASTER) &&   /* no timeout for masters */
               (now - c->lastinteraction > server.maxidletime))
          {
              !(c->flags & REDIS_MASTER) &&   /* no timeout for masters */
               (now - c->lastinteraction > server.maxidletime))
          {
-            redisLog(REDIS_DEBUG,"Closing idle client");
+            redisLog(REDIS_VERBOSE,"Closing idle client");
              freeClient(c);
          } else if (c->flags & REDIS_BLOCKED) {
              if (c->blockingto != 0 && c->blockingto < now) {
              freeClient(c);
          } else if (c->flags & REDIS_BLOCKED) {
              if (c->blockingto != 0 && c->blockingto < now) {
@@ -926,9 +1031,9 @@ static void tryResizeHashTables(void) {
  
      for (j = 0; j < server.dbnum; j++) {
          if (htNeedsResize(server.db[j].dict)) {
  
      for (j = 0; j < server.dbnum; j++) {
          if (htNeedsResize(server.db[j].dict)) {
-            redisLog(REDIS_DEBUG,"The hash table %d is too sparse, resize it...",j);
+            redisLog(REDIS_VERBOSE,"The hash table %d is too sparse, resize it...",j);
              dictResize(server.db[j].dict);
              dictResize(server.db[j].dict);
-            redisLog(REDIS_DEBUG,"Hash table %d resized.",j);
+            redisLog(REDIS_VERBOSE,"Hash table %d resized.",j);
          }
          if (htNeedsResize(server.db[j].expires))
              dictResize(server.db[j].expires);
          }
          if (htNeedsResize(server.db[j].expires))
              dictResize(server.db[j].expires);
@@ -1025,6 +1130,12 @@ static int serverCron(struct aeEventLoop *eventLoop, long long id, void *clientD
      REDIS_NOTUSED(id);
      REDIS_NOTUSED(clientData);
  
      REDIS_NOTUSED(id);
      REDIS_NOTUSED(clientData);
  
+    /* We take a cached value of the unix time in the global state because
+     * with virtual memory and aging there is to store the current time
+     * in objects at every object access, and accuracy is not needed.
+     * To access a global var is faster than calling time(NULL) */
+    server.unixtime = time(NULL);
+
      /* Update the global state with the amount of used memory */
      server.usedmemory = zmalloc_used_memory();
  
      /* Update the global state with the amount of used memory */
      server.usedmemory = zmalloc_used_memory();
  
@@ -1036,7 +1147,7 @@ static int serverCron(struct aeEventLoop *eventLoop, long long id, void *clientD
          used = dictSize(server.db[j].dict);
          vkeys = dictSize(server.db[j].expires);
          if (!(loops % 5) && (used || vkeys)) {
          used = dictSize(server.db[j].dict);
          vkeys = dictSize(server.db[j].expires);
          if (!(loops % 5) && (used || vkeys)) {
-            redisLog(REDIS_DEBUG,"DB %d: %lld keys (%lld volatile) in %lld slots HT.",j,used,vkeys,size);
+            redisLog(REDIS_VERBOSE,"DB %d: %lld keys (%lld volatile) in %lld slots HT.",j,used,vkeys,size);
              /* dictPrintStats(server.dict); */
          }
      }
              /* dictPrintStats(server.dict); */
          }
      }
@@ -1051,7 +1162,7 @@ static int serverCron(struct aeEventLoop *eventLoop, long long id, void *clientD
  
      /* Show information about connected clients */
      if (!(loops % 5)) {
  
      /* Show information about connected clients */
      if (!(loops % 5)) {
-        redisLog(REDIS_DEBUG,"%d clients connected (%d slaves), %zu bytes in use, %d shared objects",
+        redisLog(REDIS_VERBOSE,"%d clients connected (%d slaves), %zu bytes in use, %d shared objects",
              listLength(server.clients)-listLength(server.slaves),
              listLength(server.slaves),
              server.usedmemory,
              listLength(server.clients)-listLength(server.slaves),
              listLength(server.slaves),
              server.usedmemory,
@@ -1102,7 +1213,7 @@ static int serverCron(struct aeEventLoop *eventLoop, long long id, void *clientD
          /* Continue to expire if at the end of the cycle more than 25%
           * of the keys were expired. */
          do {
          /* Continue to expire if at the end of the cycle more than 25%
           * of the keys were expired. */
          do {
-            int num = dictSize(db->expires);
+            long num = dictSize(db->expires);
              time_t now = time(NULL);
  
              expired = 0;
              time_t now = time(NULL);
  
              expired = 0;
@@ -1122,6 +1233,30 @@ static int serverCron(struct aeEventLoop *eventLoop, long long id, void *clientD
          } while (expired > REDIS_EXPIRELOOKUPS_PER_CRON/4);
      }
  
          } while (expired > REDIS_EXPIRELOOKUPS_PER_CRON/4);
      }
  
+    /* Swap a few keys on disk if we are over the memory limit and VM
+     * is enbled. Try to free objects from the free list first. */
+    if (vmCanSwapOut()) {
+        while (server.vm_enabled && zmalloc_used_memory() >
+                server.vm_max_memory)
+        {
+            if (listLength(server.objfreelist)) {
+                freeOneObjectFromFreelist();
+            } else {
+                if (vmSwapOneObjectThreaded() == REDIS_ERR) {
+                    if ((loops % 30) == 0 && zmalloc_used_memory() >
+                        (server.vm_max_memory+server.vm_max_memory/10)) {
+                        redisLog(REDIS_WARNING,"WARNING: vm-max-memory limit exceeded by more than 10%% but unable to swap more objects out!");
+                    }
+                }
+                /* Note that we freed just one object, because anyway when
+                 * the I/O thread in charge to swap this object out will
+                 * do its work, the handler of completed jobs will try to swap
+                 * more objects if we are out of memory. */
+                break;
+            }
+        }
+    }
+
      /* Check if we should connect to a MASTER */
      if (server.replstate == REDIS_REPL_CONNECT) {
          redisLog(REDIS_NOTICE,"Connecting to MASTER...");
      /* Check if we should connect to a MASTER */
      if (server.replstate == REDIS_REPL_CONNECT) {
          redisLog(REDIS_NOTICE,"Connecting to MASTER...");
@@ -1185,7 +1320,7 @@ static void resetServerSaveParams() {
  static void initServerConfig() {
      server.dbnum = REDIS_DEFAULT_DBNUM;
      server.port = REDIS_SERVERPORT;
  static void initServerConfig() {
      server.dbnum = REDIS_DEFAULT_DBNUM;
      server.port = REDIS_SERVERPORT;
-    server.verbosity = REDIS_DEBUG;
+    server.verbosity = REDIS_VERBOSE;
      server.maxidletime = REDIS_MAXIDLETIME;
      server.saveparams = NULL;
      server.logfile = NULL; /* NULL = log on standard output */
      server.maxidletime = REDIS_MAXIDLETIME;
      server.saveparams = NULL;
      server.logfile = NULL; /* NULL = log on standard output */
@@ -1207,6 +1342,12 @@ static void initServerConfig() {
      server.maxclients = 0;
      server.blockedclients = 0;
      server.maxmemory = 0;
      server.maxclients = 0;
      server.blockedclients = 0;
      server.maxmemory = 0;
+    server.vm_enabled = 0;
+    server.vm_page_size = 256;          /* 256 bytes per page */
+    server.vm_pages = 1024*1024*100;    /* 104 millions of pages */
+    server.vm_max_memory = 1024LL*1024*1024*1; /* 1 GB of RAM */
+    server.vm_max_threads = 4;
+
      resetServerSaveParams();
  
      appendServerSaveParams(60*60,1);  /* save after 1 hour and 1 change */
      resetServerSaveParams();
  
      appendServerSaveParams(60*60,1);  /* save after 1 hour and 1 change */
@@ -1234,6 +1375,11 @@ static void initServer() {
      signal(SIGPIPE, SIG_IGN);
      setupSigSegvAction();
  
      signal(SIGPIPE, SIG_IGN);
      setupSigSegvAction();
  
+    server.devnull = fopen("/dev/null","w");
+    if (server.devnull == NULL) {
+        redisLog(REDIS_WARNING, "Can't open /dev/null: %s", server.neterr);
+        exit(1);
+    }
      server.clients = listCreate();
      server.slaves = listCreate();
      server.monitors = listCreate();
      server.clients = listCreate();
      server.slaves = listCreate();
      server.monitors = listCreate();
@@ -1263,7 +1409,10 @@ static void initServer() {
      server.stat_numcommands = 0;
      server.stat_numconnections = 0;
      server.stat_starttime = time(NULL);
      server.stat_numcommands = 0;
      server.stat_numconnections = 0;
      server.stat_starttime = time(NULL);
+    server.unixtime = time(NULL);
      aeCreateTimeEvent(server.el, 1, serverCron, NULL, NULL);
      aeCreateTimeEvent(server.el, 1, serverCron, NULL, NULL);
+    if (aeCreateFileEvent(server.el, server.fd, AE_READABLE,
+        acceptHandler, NULL) == AE_ERR) oom("creating file event");
  
      if (server.appendonly) {
          server.appendfd = open(server.appendfilename,O_WRONLY|O_APPEND|O_CREAT,0644);
  
      if (server.appendonly) {
          server.appendfd = open(server.appendfilename,O_WRONLY|O_APPEND|O_CREAT,0644);
@@ -1273,6 +1422,8 @@ static void initServer() {
              exit(1);
          }
      }
              exit(1);
          }
      }
+
+    if (server.vm_enabled) vmInit();
  }
  
  /* Empty the whole database */
  }
  
  /* Empty the whole database */
@@ -1357,6 +1508,7 @@ static void loadServerConfig(char *filename) {
              }
          } else if (!strcasecmp(argv[0],"loglevel") && argc == 2) {
              if (!strcasecmp(argv[1],"debug")) server.verbosity = REDIS_DEBUG;
              }
          } else if (!strcasecmp(argv[0],"loglevel") && argc == 2) {
              if (!strcasecmp(argv[1],"debug")) server.verbosity = REDIS_DEBUG;
+            else if (!strcasecmp(argv[1],"verbose")) server.verbosity = REDIS_VERBOSE;
              else if (!strcasecmp(argv[1],"notice")) server.verbosity = REDIS_NOTICE;
              else if (!strcasecmp(argv[1],"warning")) server.verbosity = REDIS_WARNING;
              else {
              else if (!strcasecmp(argv[1],"notice")) server.verbosity = REDIS_NOTICE;
              else if (!strcasecmp(argv[1],"warning")) server.verbosity = REDIS_WARNING;
              else {
@@ -1439,6 +1591,18 @@ static void loadServerConfig(char *filename) {
            server.pidfile = zstrdup(argv[1]);
          } else if (!strcasecmp(argv[0],"dbfilename") && argc == 2) {
            server.dbfilename = zstrdup(argv[1]);
            server.pidfile = zstrdup(argv[1]);
          } else if (!strcasecmp(argv[0],"dbfilename") && argc == 2) {
            server.dbfilename = zstrdup(argv[1]);
+        } else if (!strcasecmp(argv[0],"vm-enabled") && argc == 2) {
+            if ((server.vm_enabled = yesnotoi(argv[1])) == -1) {
+                err = "argument must be 'yes' or 'no'"; goto loaderr;
+            }
+        } else if (!strcasecmp(argv[0],"vm-max-memory") && argc == 2) {
+            server.vm_max_memory = strtoll(argv[1], NULL, 10);
+        } else if (!strcasecmp(argv[0],"vm-page-size") && argc == 2) {
+            server.vm_page_size = strtoll(argv[1], NULL, 10);
+        } else if (!strcasecmp(argv[0],"vm-pages") && argc == 2) {
+            server.vm_pages = strtoll(argv[1], NULL, 10);
+        } else if (!strcasecmp(argv[0],"vm-max-threads") && argc == 2) {
+            server.vm_max_threads = strtoll(argv[1], NULL, 10);
          } else {
              err = "Bad directive or wrong number of arguments"; goto loaderr;
          }
          } else {
              err = "Bad directive or wrong number of arguments"; goto loaderr;
          }
@@ -1487,9 +1651,18 @@ static void freeClient(redisClient *c) {
      listRelease(c->reply);
      freeClientArgv(c);
      close(c->fd);
      listRelease(c->reply);
      freeClientArgv(c);
      close(c->fd);
+    /* Remove from the list of clients */
      ln = listSearchKey(server.clients,c);
      redisAssert(ln != NULL);
      listDelNode(server.clients,ln);
      ln = listSearchKey(server.clients,c);
      redisAssert(ln != NULL);
      listDelNode(server.clients,ln);
+    /* Remove from the list of clients waiting for VM operations */
+    if (server.vm_enabled && listLength(c->io_keys)) {
+        ln = listSearchKey(server.io_clients,c);
+        if (ln) listDelNode(server.io_clients,ln);
+        listRelease(c->io_keys);
+    }
+    listRelease(c->io_keys);
+    /* Other cleanup */
      if (c->flags & REDIS_SLAVE) {
          if (c->replstate == REDIS_REPL_SEND_BULK && c->repldbfd != -1)
              close(c->repldbfd);
      if (c->flags & REDIS_SLAVE) {
          if (c->replstate == REDIS_REPL_SEND_BULK && c->repldbfd != -1)
              close(c->repldbfd);
@@ -1588,7 +1761,7 @@ static void sendReplyToClient(aeEventLoop *el, int fd, void *privdata, int mask)
          if (errno == EAGAIN) {
              nwritten = 0;
          } else {
          if (errno == EAGAIN) {
              nwritten = 0;
          } else {
-            redisLog(REDIS_DEBUG,
+            redisLog(REDIS_VERBOSE,
                  "Error writing to client: %s", strerror(errno));
              freeClient(c);
              return;
                  "Error writing to client: %s", strerror(errno));
              freeClient(c);
              return;
@@ -1641,7 +1814,7 @@ static void sendReplyToClientWritev(aeEventLoop *el, int fd, void *privdata, int
          /* write all collected blocks at once */
          if((nwritten = writev(fd, iov, ion)) < 0) {
              if (errno != EAGAIN) {
          /* write all collected blocks at once */
          if((nwritten = writev(fd, iov, ion)) < 0) {
              if (errno != EAGAIN) {
-                redisLog(REDIS_DEBUG,
+                redisLog(REDIS_VERBOSE,
                           "Error writing to client: %s", strerror(errno));
                  freeClient(c);
                  return;
                           "Error writing to client: %s", strerror(errno));
                  freeClient(c);
                  return;
@@ -1953,7 +2126,7 @@ again:
       * would not be called at all, but after the execution of the first commands
       * in the input buffer the client may be blocked, and the "goto again"
       * will try to reiterate. The following line will make it return asap. */
       * would not be called at all, but after the execution of the first commands
       * in the input buffer the client may be blocked, and the "goto again"
       * will try to reiterate. The following line will make it return asap. */
-    if (c->flags & REDIS_BLOCKED) return;
+    if (c->flags & REDIS_BLOCKED || c->flags & REDIS_IO_WAIT) return;
      if (c->bulklen == -1) {
          /* Read the first line of the query */
          char *p = strchr(c->querybuf,'\n');
      if (c->bulklen == -1) {
          /* Read the first line of the query */
          char *p = strchr(c->querybuf,'\n');
@@ -2002,7 +2175,7 @@ again:
              }
              return;
          } else if (sdslen(c->querybuf) >= REDIS_REQUEST_MAX_SIZE) {
              }
              return;
          } else if (sdslen(c->querybuf) >= REDIS_REQUEST_MAX_SIZE) {
-            redisLog(REDIS_DEBUG, "Client protocol error");
+            redisLog(REDIS_VERBOSE, "Client protocol error");
              freeClient(c);
              return;
          }
              freeClient(c);
              return;
          }
@@ -2039,12 +2212,12 @@ static void readQueryFromClient(aeEventLoop *el, int fd, void *privdata, int mas
          if (errno == EAGAIN) {
              nread = 0;
          } else {
          if (errno == EAGAIN) {
              nread = 0;
          } else {
-            redisLog(REDIS_DEBUG, "Reading from client: %s",strerror(errno));
+            redisLog(REDIS_VERBOSE, "Reading from client: %s",strerror(errno));
              freeClient(c);
              return;
          }
      } else if (nread == 0) {
              freeClient(c);
              return;
          }
      } else if (nread == 0) {
-        redisLog(REDIS_DEBUG, "Client closed connection");
+        redisLog(REDIS_VERBOSE, "Client closed connection");
          freeClient(c);
          return;
      }
          freeClient(c);
          return;
      }
@@ -2090,10 +2263,12 @@ static redisClient *createClient(int fd) {
      c->authenticated = 0;
      c->replstate = REDIS_REPL_NONE;
      c->reply = listCreate();
      c->authenticated = 0;
      c->replstate = REDIS_REPL_NONE;
      c->reply = listCreate();
-    c->blockingkeys = NULL;
-    c->blockingkeysnum = 0;
      listSetFreeMethod(c->reply,decrRefCount);
      listSetDupMethod(c->reply,dupClientReplyValue);
      listSetFreeMethod(c->reply,decrRefCount);
      listSetDupMethod(c->reply,dupClientReplyValue);
+    c->blockingkeys = NULL;
+    c->blockingkeysnum = 0;
+    c->io_keys = listCreate();
+    listSetFreeMethod(c->io_keys,decrRefCount);
      if (aeCreateFileEvent(server.el, c->fd, AE_READABLE,
          readQueryFromClient, c) == AE_ERR) {
          freeClient(c);
      if (aeCreateFileEvent(server.el, c->fd, AE_READABLE,
          readQueryFromClient, c) == AE_ERR) {
          freeClient(c);
@@ -2110,6 +2285,11 @@ static void addReply(redisClient *c, robj *obj) {
           c->replstate == REDIS_REPL_ONLINE) &&
          aeCreateFileEvent(server.el, c->fd, AE_WRITABLE,
          sendReplyToClient, c) == AE_ERR) return;
           c->replstate == REDIS_REPL_ONLINE) &&
          aeCreateFileEvent(server.el, c->fd, AE_WRITABLE,
          sendReplyToClient, c) == AE_ERR) return;
+
+    if (server.vm_enabled && obj->storage != REDIS_VM_MEMORY) {
+        obj = dupStringObject(obj);
+        obj->refcount = 0; /* getDecodedObject() will increment the refcount */
+    }
      listAddNodeTail(c->reply,getDecodedObject(obj));
  }
  
      listAddNodeTail(c->reply,getDecodedObject(obj));
  }
  
@@ -2158,10 +2338,10 @@ static void acceptHandler(aeEventLoop *el, int fd, void *privdata, int mask) {
  
      cfd = anetAccept(server.neterr, fd, cip, &cport);
      if (cfd == AE_ERR) {
  
      cfd = anetAccept(server.neterr, fd, cip, &cport);
      if (cfd == AE_ERR) {
-        redisLog(REDIS_DEBUG,"Accepting client connection: %s", server.neterr);
+        redisLog(REDIS_VERBOSE,"Accepting client connection: %s", server.neterr);
          return;
      }
          return;
      }
-    redisLog(REDIS_DEBUG,"Accepted %s:%d", cip, cport);
+    redisLog(REDIS_VERBOSE,"Accepted %s:%d", cip, cport);
      if ((c = createClient(cfd)) == NULL) {
          redisLog(REDIS_WARNING,"Error allocating resoures for the client");
          close(cfd); /* May be already closed, just ingore errors */
      if ((c = createClient(cfd)) == NULL) {
          redisLog(REDIS_WARNING,"Error allocating resoures for the client");
          close(cfd); /* May be already closed, just ingore errors */
@@ -2194,12 +2374,20 @@ static robj *createObject(int type, void *ptr) {
          o = listNodeValue(head);
          listDelNode(server.objfreelist,head);
      } else {
          o = listNodeValue(head);
          listDelNode(server.objfreelist,head);
      } else {
-        o = zmalloc(sizeof(*o));
+        if (server.vm_enabled) {
+            o = zmalloc(sizeof(*o));
+        } else {
+            o = zmalloc(sizeof(*o)-sizeof(struct redisObjectVM));
+        }
      }
      o->type = type;
      o->encoding = REDIS_ENCODING_RAW;
      o->ptr = ptr;
      o->refcount = 1;
      }
      o->type = type;
      o->encoding = REDIS_ENCODING_RAW;
      o->ptr = ptr;
      o->refcount = 1;
+    if (server.vm_enabled) {
+        o->vm.atime = server.unixtime;
+        o->storage = REDIS_VM_MEMORY;
+    }
      return o;
  }
  
      return o;
  }
  
@@ -2207,6 +2395,11 @@ static robj *createStringObject(char *ptr, size_t len) {
      return createObject(REDIS_STRING,sdsnewlen(ptr,len));
  }
  
      return createObject(REDIS_STRING,sdsnewlen(ptr,len));
  }
  
+static robj *dupStringObject(robj *o) {
+    assert(o->encoding == REDIS_ENCODING_RAW);
+    return createStringObject(o->ptr,sdslen(o->ptr));
+}
+
  static robj *createListObject(void) {
      list *l = listCreate();
  
  static robj *createListObject(void) {
      list *l = listCreate();
  
@@ -2254,21 +2447,34 @@ static void freeHashObject(robj *o) {
  }
  
  static void incrRefCount(robj *o) {
  }
  
  static void incrRefCount(robj *o) {
+    redisAssert(!server.vm_enabled || o->storage == REDIS_VM_MEMORY);
      o->refcount++;
      o->refcount++;
-#ifdef DEBUG_REFCOUNT
-    if (o->type == REDIS_STRING)
-        printf("Increment '%s'(%p), now is: %d\n",o->ptr,o,o->refcount);
-#endif
  }
  
  static void decrRefCount(void *obj) {
      robj *o = obj;
  
  }
  
  static void decrRefCount(void *obj) {
      robj *o = obj;
  
-#ifdef DEBUG_REFCOUNT
-    if (o->type == REDIS_STRING)
-        printf("Decrement '%s'(%p), now is: %d\n",o->ptr,o,o->refcount-1);
-#endif
+    /* Object is swapped out, or in the process of being loaded. */
+    if (server.vm_enabled &&
+        (o->storage == REDIS_VM_SWAPPED || o->storage == REDIS_VM_LOADING))
+    {
+        if (o->storage == REDIS_VM_SWAPPED || o->storage == REDIS_VM_LOADING) {
+            redisAssert(o->refcount == 1);
+        }
+        if (o->storage == REDIS_VM_LOADING) vmCancelThreadedIOJob(obj);
+        redisAssert(o->type == REDIS_STRING);
+        freeStringObject(o);
+        vmMarkPagesFree(o->vm.page,o->vm.usedpages);
+        if (listLength(server.objfreelist) > REDIS_OBJFREELIST_MAX ||
+            !listAddNodeHead(server.objfreelist,o))
+            zfree(o);
+        server.vm_stats_swapped_objects--;
+        return;
+    }
+    /* Object is in memory, or in the process of being swapped out. */
      if (--(o->refcount) == 0) {
      if (--(o->refcount) == 0) {
+        if (server.vm_enabled && o->storage == REDIS_VM_SWAPPING)
+            vmCancelThreadedIOJob(obj);
          switch(o->type) {
          case REDIS_STRING: freeStringObject(o); break;
          case REDIS_LIST: freeListObject(o); break;
          switch(o->type) {
          case REDIS_STRING: freeStringObject(o); break;
          case REDIS_LIST: freeListObject(o); break;
@@ -2285,7 +2491,31 @@ static void decrRefCount(void *obj) {
  
  static robj *lookupKey(redisDb *db, robj *key) {
      dictEntry *de = dictFind(db->dict,key);
  
  static robj *lookupKey(redisDb *db, robj *key) {
      dictEntry *de = dictFind(db->dict,key);
-    return de ? dictGetEntryVal(de) : NULL;
+    if (de) {
+        robj *key = dictGetEntryKey(de);
+        robj *val = dictGetEntryVal(de);
+
+        if (server.vm_enabled) {
+            if (key->storage == REDIS_VM_MEMORY ||
+                key->storage == REDIS_VM_SWAPPING)
+            {
+                /* If we were swapping the object out, stop it, this key
+                 * was requested. */
+                if (key->storage == REDIS_VM_SWAPPING)
+                    vmCancelThreadedIOJob(key);
+                /* Update the access time of the key for the aging algorithm. */
+                key->vm.atime = server.unixtime;
+            } else {
+                /* Our value was swapped on disk. Bring it at home. */
+                redisAssert(val == NULL);
+                val = vmLoadObject(key);
+                dictGetEntryVal(de) = val;
+            }
+        }
+        return val;
+    } else {
+        return NULL;
+    }
  }
  
  static robj *lookupKeyRead(redisDb *db, robj *key) {
  }
  
  static robj *lookupKeyRead(redisDb *db, robj *key) {
@@ -2468,7 +2698,7 @@ static size_t stringObjectLen(robj *o) {
      }
  }
  
      }
  }
  
-/*============================ DB saving/loading ============================ */
+/*============================ RDB saving/loading =========================== */
  
  static int rdbSaveType(FILE *fp, unsigned char type) {
      if (fwrite(&type,1,1,fp) == 0) return -1;
  
  static int rdbSaveType(FILE *fp, unsigned char type) {
      if (fwrite(&type,1,1,fp) == 0) return -1;
@@ -2608,9 +2838,23 @@ static int rdbSaveStringObjectRaw(FILE *fp, robj *obj) {
  static int rdbSaveStringObject(FILE *fp, robj *obj) {
      int retval;
  
  static int rdbSaveStringObject(FILE *fp, robj *obj) {
      int retval;
  
-    obj = getDecodedObject(obj);
-    retval = rdbSaveStringObjectRaw(fp,obj);
-    decrRefCount(obj);
+    if (obj->storage == REDIS_VM_MEMORY &&
+       obj->encoding != REDIS_ENCODING_RAW)
+    {
+        obj = getDecodedObject(obj);
+        retval = rdbSaveStringObjectRaw(fp,obj);
+        decrRefCount(obj);
+    } else {
+        /* This is a fast path when we are sure the object is not encoded.
+         * Note that's any *faster* actually as we needed to add the conditional
+         * but because this may happen in a background process we don't want
+         * to touch the object fields with incr/decrRefCount in order to
+         * preveny copy on write of pages.
+         *
+         * Also incrRefCount() will have a failing assert() if we try to call
+         * it against an object with storage != REDIS_VM_MEMORY. */
+        retval = rdbSaveStringObjectRaw(fp,obj);
+    }
      return retval;
  }
  
      return retval;
  }
  
@@ -2641,6 +2885,75 @@ static int rdbSaveDoubleValue(FILE *fp, double val) {
      return 0;
  }
  
      return 0;
  }
  
+/* Save a Redis object. */
+static int rdbSaveObject(FILE *fp, robj *o) {
+    if (o->type == REDIS_STRING) {
+        /* Save a string value */
+        if (rdbSaveStringObject(fp,o) == -1) return -1;
+    } else if (o->type == REDIS_LIST) {
+        /* Save a list value */
+        list *list = o->ptr;
+        listNode *ln;
+
+        listRewind(list);
+        if (rdbSaveLen(fp,listLength(list)) == -1) return -1;
+        while((ln = listYield(list))) {
+            robj *eleobj = listNodeValue(ln);
+
+            if (rdbSaveStringObject(fp,eleobj) == -1) return -1;
+        }
+    } else if (o->type == REDIS_SET) {
+        /* Save a set value */
+        dict *set = o->ptr;
+        dictIterator *di = dictGetIterator(set);
+        dictEntry *de;
+
+        if (rdbSaveLen(fp,dictSize(set)) == -1) return -1;
+        while((de = dictNext(di)) != NULL) {
+            robj *eleobj = dictGetEntryKey(de);
+
+            if (rdbSaveStringObject(fp,eleobj) == -1) return -1;
+        }
+        dictReleaseIterator(di);
+    } else if (o->type == REDIS_ZSET) {
+        /* Save a set value */
+        zset *zs = o->ptr;
+        dictIterator *di = dictGetIterator(zs->dict);
+        dictEntry *de;
+
+        if (rdbSaveLen(fp,dictSize(zs->dict)) == -1) return -1;
+        while((de = dictNext(di)) != NULL) {
+            robj *eleobj = dictGetEntryKey(de);
+            double *score = dictGetEntryVal(de);
+
+            if (rdbSaveStringObject(fp,eleobj) == -1) return -1;
+            if (rdbSaveDoubleValue(fp,*score) == -1) return -1;
+        }
+        dictReleaseIterator(di);
+    } else {
+        redisAssert(0 != 0);
+    }
+    return 0;
+}
+
+/* Return the length the object will have on disk if saved with
+ * the rdbSaveObject() function. Currently we use a trick to get
+ * this length with very little changes to the code. In the future
+ * we could switch to a faster solution. */
+static off_t rdbSavedObjectLen(robj *o, FILE *fp) {
+    if (fp == NULL) fp = server.devnull;
+    rewind(fp);
+    assert(rdbSaveObject(fp,o) != 1);
+    return ftello(fp);
+}
+
+/* Return the number of pages required to save this object in the swap file */
+static off_t rdbSavedObjectPages(robj *o, FILE *fp) {
+    off_t bytes = rdbSavedObjectLen(o,fp);
+    
+    return (bytes+(server.vm_page_size-1))/server.vm_page_size;
+}
+
  /* Save the DB on disk. Return REDIS_ERR on error, REDIS_OK on success */
  static int rdbSave(char *filename) {
      dictIterator *di = NULL;
  /* Save the DB on disk. Return REDIS_ERR on error, REDIS_OK on success */
  static int rdbSave(char *filename) {
      dictIterator *di = NULL;
@@ -2684,54 +2997,25 @@ static int rdbSave(char *filename) {
                  if (rdbSaveType(fp,REDIS_EXPIRETIME) == -1) goto werr;
                  if (rdbSaveTime(fp,expiretime) == -1) goto werr;
              }
                  if (rdbSaveType(fp,REDIS_EXPIRETIME) == -1) goto werr;
                  if (rdbSaveTime(fp,expiretime) == -1) goto werr;
              }
-            /* Save the key and associated value */
-            if (rdbSaveType(fp,o->type) == -1) goto werr;
-            if (rdbSaveStringObject(fp,key) == -1) goto werr;
-            if (o->type == REDIS_STRING) {
-                /* Save a string value */
-                if (rdbSaveStringObject(fp,o) == -1) goto werr;
-            } else if (o->type == REDIS_LIST) {
-                /* Save a list value */
-                list *list = o->ptr;
-                listNode *ln;
-
-                listRewind(list);
-                if (rdbSaveLen(fp,listLength(list)) == -1) goto werr;
-                while((ln = listYield(list))) {
-                    robj *eleobj = listNodeValue(ln);
-
-                    if (rdbSaveStringObject(fp,eleobj) == -1) goto werr;
-                }
-            } else if (o->type == REDIS_SET) {
-                /* Save a set value */
-                dict *set = o->ptr;
-                dictIterator *di = dictGetIterator(set);
-                dictEntry *de;
-
-                if (rdbSaveLen(fp,dictSize(set)) == -1) goto werr;
-                while((de = dictNext(di)) != NULL) {
-                    robj *eleobj = dictGetEntryKey(de);
-
-                    if (rdbSaveStringObject(fp,eleobj) == -1) goto werr;
-                }
-                dictReleaseIterator(di);
-            } else if (o->type == REDIS_ZSET) {
-                /* Save a set value */
-                zset *zs = o->ptr;
-                dictIterator *di = dictGetIterator(zs->dict);
-                dictEntry *de;
-
-                if (rdbSaveLen(fp,dictSize(zs->dict)) == -1) goto werr;
-                while((de = dictNext(di)) != NULL) {
-                    robj *eleobj = dictGetEntryKey(de);
-                    double *score = dictGetEntryVal(de);
-
-                    if (rdbSaveStringObject(fp,eleobj) == -1) goto werr;
-                    if (rdbSaveDoubleValue(fp,*score) == -1) goto werr;
-                }
-                dictReleaseIterator(di);
+            /* Save the key and associated value. This requires special
+             * handling if the value is swapped out. */
+            if (!server.vm_enabled || key->storage == REDIS_VM_MEMORY ||
+                                      key->storage == REDIS_VM_SWAPPING) {
+                /* Save type, key, value */
+                if (rdbSaveType(fp,o->type) == -1) goto werr;
+                if (rdbSaveStringObject(fp,key) == -1) goto werr;
+                if (rdbSaveObject(fp,o) == -1) goto werr;
              } else {
              } else {
-                redisAssert(0 != 0);
+                /* REDIS_VM_SWAPPED or REDIS_VM_LOADING */
+                robj *po;
+                /* Get a preview of the object in memory */
+                po = vmPreviewObject(key);
+                /* Save type, key, value */
+                if (rdbSaveType(fp,key->vtype) == -1) goto werr;
+                if (rdbSaveStringObject(fp,key) == -1) goto werr;
+                if (rdbSaveObject(fp,po) == -1) goto werr;
+                /* Remove the loaded object from memory */
+                decrRefCount(po);
              }
          }
          dictReleaseIterator(di);
              }
          }
          dictReleaseIterator(di);
@@ -2814,35 +3098,29 @@ static time_t rdbLoadTime(FILE *fp) {
   *
   * isencoded is set to 1 if the readed length is not actually a length but
   * an "encoding type", check the above comments for more info */
   *
   * isencoded is set to 1 if the readed length is not actually a length but
   * an "encoding type", check the above comments for more info */
-static uint32_t rdbLoadLen(FILE *fp, int rdbver, int *isencoded) {
+static uint32_t rdbLoadLen(FILE *fp, int *isencoded) {
      unsigned char buf[2];
      uint32_t len;
      unsigned char buf[2];
      uint32_t len;
+    int type;
  
      if (isencoded) *isencoded = 0;
  
      if (isencoded) *isencoded = 0;
-    if (rdbver == 0) {
+    if (fread(buf,1,1,fp) == 0) return REDIS_RDB_LENERR;
+    type = (buf[0]&0xC0)>>6;
+    if (type == REDIS_RDB_6BITLEN) {
+        /* Read a 6 bit len */
+        return buf[0]&0x3F;
+    } else if (type == REDIS_RDB_ENCVAL) {
+        /* Read a 6 bit len encoding type */
+        if (isencoded) *isencoded = 1;
+        return buf[0]&0x3F;
+    } else if (type == REDIS_RDB_14BITLEN) {
+        /* Read a 14 bit len */
+        if (fread(buf+1,1,1,fp) == 0) return REDIS_RDB_LENERR;
+        return ((buf[0]&0x3F)<<8)|buf[1];
+    } else {
+        /* Read a 32 bit len */
          if (fread(&len,4,1,fp) == 0) return REDIS_RDB_LENERR;
          return ntohl(len);
          if (fread(&len,4,1,fp) == 0) return REDIS_RDB_LENERR;
          return ntohl(len);
-    } else {
-        int type;
-
-        if (fread(buf,1,1,fp) == 0) return REDIS_RDB_LENERR;
-        type = (buf[0]&0xC0)>>6;
-        if (type == REDIS_RDB_6BITLEN) {
-            /* Read a 6 bit len */
-            return buf[0]&0x3F;
-        } else if (type == REDIS_RDB_ENCVAL) {
-            /* Read a 6 bit len encoding type */
-            if (isencoded) *isencoded = 1;
-            return buf[0]&0x3F;
-        } else if (type == REDIS_RDB_14BITLEN) {
-            /* Read a 14 bit len */
-            if (fread(buf+1,1,1,fp) == 0) return REDIS_RDB_LENERR;
-            return ((buf[0]&0x3F)<<8)|buf[1];
-        } else {
-            /* Read a 32 bit len */
-            if (fread(&len,4,1,fp) == 0) return REDIS_RDB_LENERR;
-            return ntohl(len);
-        }
      }
  }
  
      }
  }
  
@@ -2870,13 +3148,13 @@ static robj *rdbLoadIntegerObject(FILE *fp, int enctype) {
      return createObject(REDIS_STRING,sdscatprintf(sdsempty(),"%lld",val));
  }
  
      return createObject(REDIS_STRING,sdscatprintf(sdsempty(),"%lld",val));
  }
  
-static robj *rdbLoadLzfStringObject(FILE*fp, int rdbver) {
+static robj *rdbLoadLzfStringObject(FILE*fp) {
      unsigned int len, clen;
      unsigned char *c = NULL;
      sds val = NULL;
  
      unsigned int len, clen;
      unsigned char *c = NULL;
      sds val = NULL;
  
-    if ((clen = rdbLoadLen(fp,rdbver,NULL)) == REDIS_RDB_LENERR) return NULL;
-    if ((len = rdbLoadLen(fp,rdbver,NULL)) == REDIS_RDB_LENERR) return NULL;
+    if ((clen = rdbLoadLen(fp,NULL)) == REDIS_RDB_LENERR) return NULL;
+    if ((len = rdbLoadLen(fp,NULL)) == REDIS_RDB_LENERR) return NULL;
      if ((c = zmalloc(clen)) == NULL) goto err;
      if ((val = sdsnewlen(NULL,len)) == NULL) goto err;
      if (fread(c,clen,1,fp) == 0) goto err;
      if ((c = zmalloc(clen)) == NULL) goto err;
      if ((val = sdsnewlen(NULL,len)) == NULL) goto err;
      if (fread(c,clen,1,fp) == 0) goto err;
@@ -2889,12 +3167,12 @@ err:
      return NULL;
  }
  
      return NULL;
  }
  
-static robj *rdbLoadStringObject(FILE*fp, int rdbver) {
+static robj *rdbLoadStringObject(FILE*fp) {
      int isencoded;
      uint32_t len;
      sds val;
  
      int isencoded;
      uint32_t len;
      sds val;
  
-    len = rdbLoadLen(fp,rdbver,&isencoded);
+    len = rdbLoadLen(fp,&isencoded);
      if (isencoded) {
          switch(len) {
          case REDIS_RDB_ENC_INT8:
      if (isencoded) {
          switch(len) {
          case REDIS_RDB_ENC_INT8:
@@ -2902,7 +3180,7 @@ static robj *rdbLoadStringObject(FILE*fp, int rdbver) {
          case REDIS_RDB_ENC_INT32:
              return tryObjectSharing(rdbLoadIntegerObject(fp,len));
          case REDIS_RDB_ENC_LZF:
          case REDIS_RDB_ENC_INT32:
              return tryObjectSharing(rdbLoadIntegerObject(fp,len));
          case REDIS_RDB_ENC_LZF:
-            return tryObjectSharing(rdbLoadLzfStringObject(fp,rdbver));
+            return tryObjectSharing(rdbLoadLzfStringObject(fp));
          default:
              redisAssert(0!=0);
          }
          default:
              redisAssert(0!=0);
          }
@@ -2935,6 +3213,59 @@ static int rdbLoadDoubleValue(FILE *fp, double *val) {
      }
  }
  
      }
  }
  
+/* Load a Redis object of the specified type from the specified file.
+ * On success a newly allocated object is returned, otherwise NULL. */
+static robj *rdbLoadObject(int type, FILE *fp) {
+    robj *o;
+
+    if (type == REDIS_STRING) {
+        /* Read string value */
+        if ((o = rdbLoadStringObject(fp)) == NULL) return NULL;
+        tryObjectEncoding(o);
+    } else if (type == REDIS_LIST || type == REDIS_SET) {
+        /* Read list/set value */
+        uint32_t listlen;
+
+        if ((listlen = rdbLoadLen(fp,NULL)) == REDIS_RDB_LENERR) return NULL;
+        o = (type == REDIS_LIST) ? createListObject() : createSetObject();
+        /* Load every single element of the list/set */
+        while(listlen--) {
+            robj *ele;
+
+            if ((ele = rdbLoadStringObject(fp)) == NULL) return NULL;
+            tryObjectEncoding(ele);
+            if (type == REDIS_LIST) {
+                listAddNodeTail((list*)o->ptr,ele);
+            } else {
+                dictAdd((dict*)o->ptr,ele,NULL);
+            }
+        }
+    } else if (type == REDIS_ZSET) {
+        /* Read list/set value */
+        uint32_t zsetlen;
+        zset *zs;
+
+        if ((zsetlen = rdbLoadLen(fp,NULL)) == REDIS_RDB_LENERR) return NULL;
+        o = createZsetObject();
+        zs = o->ptr;
+        /* Load every single element of the list/set */
+        while(zsetlen--) {
+            robj *ele;
+            double *score = zmalloc(sizeof(double));
+
+            if ((ele = rdbLoadStringObject(fp)) == NULL) return NULL;
+            tryObjectEncoding(ele);
+            if (rdbLoadDoubleValue(fp,score) == -1) return NULL;
+            dictAdd(zs->dict,ele,score);
+            zslInsert(zs->zsl,*score,ele);
+            incrRefCount(ele); /* added to skiplist */
+        }
+    } else {
+        redisAssert(0 != 0);
+    }
+    return o;
+}
+
  static int rdbLoad(char *filename) {
      FILE *fp;
      robj *keyobj = NULL;
  static int rdbLoad(char *filename) {
      FILE *fp;
      robj *keyobj = NULL;
@@ -2944,6 +3275,7 @@ static int rdbLoad(char *filename) {
      redisDb *db = server.db+0;
      char buf[1024];
      time_t expiretime = -1, now = time(NULL);
      redisDb *db = server.db+0;
      char buf[1024];
      time_t expiretime = -1, now = time(NULL);
+    long long loadedkeys = 0;
  
      fp = fopen(filename,"r");
      if (!fp) return REDIS_ERR;
  
      fp = fopen(filename,"r");
      if (!fp) return REDIS_ERR;
@@ -2955,7 +3287,7 @@ static int rdbLoad(char *filename) {
          return REDIS_ERR;
      }
      rdbver = atoi(buf+5);
          return REDIS_ERR;
      }
      rdbver = atoi(buf+5);
-    if (rdbver > 1) {
+    if (rdbver != 1) {
          fclose(fp);
          redisLog(REDIS_WARNING,"Can't handle RDB format version %d",rdbver);
          return REDIS_ERR;
          fclose(fp);
          redisLog(REDIS_WARNING,"Can't handle RDB format version %d",rdbver);
          return REDIS_ERR;
@@ -2973,7 +3305,7 @@ static int rdbLoad(char *filename) {
          if (type == REDIS_EOF) break;
          /* Handle SELECT DB opcode as a special case */
          if (type == REDIS_SELECTDB) {
          if (type == REDIS_EOF) break;
          /* Handle SELECT DB opcode as a special case */
          if (type == REDIS_SELECTDB) {
-            if ((dbid = rdbLoadLen(fp,rdbver,NULL)) == REDIS_RDB_LENERR)
+            if ((dbid = rdbLoadLen(fp,NULL)) == REDIS_RDB_LENERR)
                  goto eoferr;
              if (dbid >= (unsigned)server.dbnum) {
                  redisLog(REDIS_WARNING,"FATAL: Data file was created with a Redis server configured to handle more than %d databases. Exiting\n", server.dbnum);
                  goto eoferr;
              if (dbid >= (unsigned)server.dbnum) {
                  redisLog(REDIS_WARNING,"FATAL: Data file was created with a Redis server configured to handle more than %d databases. Exiting\n", server.dbnum);
@@ -2984,55 +3316,9 @@ static int rdbLoad(char *filename) {
              continue;
          }
          /* Read key */
              continue;
          }
          /* Read key */
-        if ((keyobj = rdbLoadStringObject(fp,rdbver)) == NULL) goto eoferr;
-
-        if (type == REDIS_STRING) {
-            /* Read string value */
-            if ((o = rdbLoadStringObject(fp,rdbver)) == NULL) goto eoferr;
-            tryObjectEncoding(o);
-        } else if (type == REDIS_LIST || type == REDIS_SET) {
-            /* Read list/set value */
-            uint32_t listlen;
-
-            if ((listlen = rdbLoadLen(fp,rdbver,NULL)) == REDIS_RDB_LENERR)
-                goto eoferr;
-            o = (type == REDIS_LIST) ? createListObject() : createSetObject();
-            /* Load every single element of the list/set */
-            while(listlen--) {
-                robj *ele;
-
-                if ((ele = rdbLoadStringObject(fp,rdbver)) == NULL) goto eoferr;
-                tryObjectEncoding(ele);
-                if (type == REDIS_LIST) {
-                    listAddNodeTail((list*)o->ptr,ele);
-                } else {
-                    dictAdd((dict*)o->ptr,ele,NULL);
-                }
-            }
-        } else if (type == REDIS_ZSET) {
-            /* Read list/set value */
-            uint32_t zsetlen;
-            zset *zs;
-
-            if ((zsetlen = rdbLoadLen(fp,rdbver,NULL)) == REDIS_RDB_LENERR)
-                goto eoferr;
-            o = createZsetObject();
-            zs = o->ptr;
-            /* Load every single element of the list/set */
-            while(zsetlen--) {
-                robj *ele;
-                double *score = zmalloc(sizeof(double));
-
-                if ((ele = rdbLoadStringObject(fp,rdbver)) == NULL) goto eoferr;
-                tryObjectEncoding(ele);
-                if (rdbLoadDoubleValue(fp,score) == -1) goto eoferr;
-                dictAdd(zs->dict,ele,score);
-                zslInsert(zs->zsl,*score,ele);
-                incrRefCount(ele); /* added to skiplist */
-            }
-        } else {
-            redisAssert(0 != 0);
-        }
+        if ((keyobj = rdbLoadStringObject(fp)) == NULL) goto eoferr;
+        /* Read value */
+        if ((o = rdbLoadObject(type,fp)) == NULL) goto eoferr;
          /* Add the new object in the hash table */
          retval = dictAdd(d,keyobj,o);
          if (retval == DICT_ERR) {
          /* Add the new object in the hash table */
          retval = dictAdd(d,keyobj,o);
          if (retval == DICT_ERR) {
@@ -3047,6 +3333,13 @@ static int rdbLoad(char *filename) {
              expiretime = -1;
          }
          keyobj = o = NULL;
              expiretime = -1;
          }
          keyobj = o = NULL;
+        /* Handle swapping while loading big datasets when VM is on */
+        loadedkeys++;
+        if (server.vm_enabled && (loadedkeys % 5000) == 0) {
+            while (zmalloc_used_memory() > server.vm_max_memory) {
+                if (vmSwapOneObjectBlocking() == REDIS_ERR) break;
+            }
+        }
      }
      fclose(fp);
      return REDIS_OK;
      }
      fclose(fp);
      return REDIS_OK;
@@ -3089,6 +3382,12 @@ static void setGenericCommand(redisClient *c, int nx) {
      retval = dictAdd(c->db->dict,c->argv[1],c->argv[2]);
      if (retval == DICT_ERR) {
          if (!nx) {
      retval = dictAdd(c->db->dict,c->argv[1],c->argv[2]);
      if (retval == DICT_ERR) {
          if (!nx) {
+            /* If the key is about a swapped value, we want a new key object
+             * to overwrite the old. So we delete the old key in the database.
+             * This will also make sure that swap pages about the old object
+             * will be marked as free. */
+            if (deleteIfSwapped(c->db,c->argv[1]))
+                incrRefCount(c->argv[1]);
              dictReplace(c->db->dict,c->argv[1],c->argv[2]);
              incrRefCount(c->argv[2]);
          } else {
              dictReplace(c->db->dict,c->argv[1],c->argv[2]);
              incrRefCount(c->argv[2]);
          } else {
@@ -3862,20 +4161,24 @@ static void rpoplpushcommand(redisClient *c) {
                  robj *ele = listNodeValue(ln);
                  list *dstlist;
  
                  robj *ele = listNodeValue(ln);
                  list *dstlist;
  
-                if (dobj == NULL) {
-
-                    /* Create the list if the key does not exist */
-                    dobj = createListObject();
-                    dictAdd(c->db->dict,c->argv[2],dobj);
-                    incrRefCount(c->argv[2]);
-                } else if (dobj->type != REDIS_LIST) {
+                if (dobj && dobj->type != REDIS_LIST) {
                      addReply(c,shared.wrongtypeerr);
                      return;
                  }
                      addReply(c,shared.wrongtypeerr);
                      return;
                  }
-                /* Add the element to the target list */
-                dstlist = dobj->ptr;
-                listAddNodeHead(dstlist,ele);
-                incrRefCount(ele);
+
+                /* Add the element to the target list (unless it's directly
+                 * passed to some BLPOP-ing client */
+                if (!handleClientsWaitingListPush(c,c->argv[2],ele)) {
+                    if (dobj == NULL) {
+                        /* Create the list if the key does not exist */
+                        dobj = createListObject();
+                        dictAdd(c->db->dict,c->argv[2],dobj);
+                        incrRefCount(c->argv[2]);
+                    }
+                    dstlist = dobj->ptr;
+                    listAddNodeHead(dstlist,ele);
+                    incrRefCount(ele);
+                }
  
                  /* Send the element to the client as reply as well */
                  addReplyBulkLen(c,ele);
  
                  /* Send the element to the client as reply as well */
                  addReplyBulkLen(c,ele);
@@ -5201,6 +5504,27 @@ static void sortCommand(redisClient *c) {
      zfree(vector);
  }
  
      zfree(vector);
  }
  
+/* Convert an amount of bytes into a human readable string in the form
+ * of 100B, 2G, 100M, 4K, and so forth. */
+static void bytesToHuman(char *s, unsigned long long n) {
+    double d;
+
+    if (n < 1024) {
+        /* Bytes */
+        sprintf(s,"%lluB",n);
+        return;
+    } else if (n < (1024*1024)) {
+        d = (double)n/(1024);
+        sprintf(s,"%.2fK",d);
+    } else if (n < (1024LL*1024*1024)) {
+        d = (double)n/(1024*1024);
+        sprintf(s,"%.2fM",d);
+    } else if (n < (1024LL*1024*1024*1024)) {
+        d = (double)n/(1024LL*1024*1024);
+        sprintf(s,"%.2fM",d);
+    }
+}
+
  /* Create the string returned by the INFO command. This is decoupled
   * by the INFO command itself as we need to report the same information
   * on memory corruption problems. */
  /* Create the string returned by the INFO command. This is decoupled
   * by the INFO command itself as we need to report the same information
   * on memory corruption problems. */
@@ -5208,39 +5532,47 @@ static sds genRedisInfoString(void) {
      sds info;
      time_t uptime = time(NULL)-server.stat_starttime;
      int j;
      sds info;
      time_t uptime = time(NULL)-server.stat_starttime;
      int j;
-    
+    char hmem[64];
+  
+    bytesToHuman(hmem,server.usedmemory);
      info = sdscatprintf(sdsempty(),
          "redis_version:%s\r\n"
          "arch_bits:%s\r\n"
          "multiplexing_api:%s\r\n"
      info = sdscatprintf(sdsempty(),
          "redis_version:%s\r\n"
          "arch_bits:%s\r\n"
          "multiplexing_api:%s\r\n"
+        "process_id:%ld\r\n"
          "uptime_in_seconds:%ld\r\n"
          "uptime_in_days:%ld\r\n"
          "connected_clients:%d\r\n"
          "connected_slaves:%d\r\n"
          "blocked_clients:%d\r\n"
          "used_memory:%zu\r\n"
          "uptime_in_seconds:%ld\r\n"
          "uptime_in_days:%ld\r\n"
          "connected_clients:%d\r\n"
          "connected_slaves:%d\r\n"
          "blocked_clients:%d\r\n"
          "used_memory:%zu\r\n"
+        "used_memory_human:%s\r\n"
          "changes_since_last_save:%lld\r\n"
          "bgsave_in_progress:%d\r\n"
          "last_save_time:%ld\r\n"
          "bgrewriteaof_in_progress:%d\r\n"
          "total_connections_received:%lld\r\n"
          "total_commands_processed:%lld\r\n"
          "changes_since_last_save:%lld\r\n"
          "bgsave_in_progress:%d\r\n"
          "last_save_time:%ld\r\n"
          "bgrewriteaof_in_progress:%d\r\n"
          "total_connections_received:%lld\r\n"
          "total_commands_processed:%lld\r\n"
+        "vm_enabled:%d\r\n"
          "role:%s\r\n"
          ,REDIS_VERSION,
          (sizeof(long) == 8) ? "64" : "32",
          aeGetApiName(),
          "role:%s\r\n"
          ,REDIS_VERSION,
          (sizeof(long) == 8) ? "64" : "32",
          aeGetApiName(),
+        (long) getpid(),
          uptime,
          uptime/(3600*24),
          listLength(server.clients)-listLength(server.slaves),
          listLength(server.slaves),
          server.blockedclients,
          server.usedmemory,
          uptime,
          uptime/(3600*24),
          listLength(server.clients)-listLength(server.slaves),
          listLength(server.slaves),
          server.blockedclients,
          server.usedmemory,
+        hmem,
          server.dirty,
          server.bgsavechildpid != -1,
          server.lastsave,
          server.bgrewritechildpid != -1,
          server.stat_numconnections,
          server.stat_numcommands,
          server.dirty,
          server.bgsavechildpid != -1,
          server.lastsave,
          server.bgrewritechildpid != -1,
          server.stat_numconnections,
          server.stat_numcommands,
+        server.vm_enabled != 0,
          server.masterhost == NULL ? "master" : "slave"
      );
      if (server.masterhost) {
          server.masterhost == NULL ? "master" : "slave"
      );
      if (server.masterhost) {
@@ -5256,6 +5588,32 @@ static sds genRedisInfoString(void) {
              server.master ? ((int)(time(NULL)-server.master->lastinteraction)) : -1
          );
      }
              server.master ? ((int)(time(NULL)-server.master->lastinteraction)) : -1
          );
      }
+    if (server.vm_enabled) {
+        info = sdscatprintf(info,
+            "vm_conf_max_memory:%llu\r\n"
+            "vm_conf_page_size:%llu\r\n"
+            "vm_conf_pages:%llu\r\n"
+            "vm_stats_used_pages:%llu\r\n"
+            "vm_stats_swapped_objects:%llu\r\n"
+            "vm_stats_swappin_count:%llu\r\n"
+            "vm_stats_swappout_count:%llu\r\n"
+            "vm_stats_io_newjobs_len:%lu\r\n"
+            "vm_stats_io_processing_len:%lu\r\n"
+            "vm_stats_io_processed_len:%lu\r\n"
+            "vm_stats_io_waiting_clients:%lu\r\n"
+            ,(unsigned long long) server.vm_max_memory,
+            (unsigned long long) server.vm_page_size,
+            (unsigned long long) server.vm_pages,
+            (unsigned long long) server.vm_stats_used_pages,
+            (unsigned long long) server.vm_stats_swapped_objects,
+            (unsigned long long) server.vm_stats_swapins,
+            (unsigned long long) server.vm_stats_swapouts,
+            (unsigned long) listLength(server.io_newjobs),
+            (unsigned long) listLength(server.io_processing),
+            (unsigned long) listLength(server.io_processed),
+            (unsigned long) listLength(server.io_clients)
+        );
+    }
      for (j = 0; j < server.dbnum; j++) {
          long long keys, vkeys;
  
      for (j = 0; j < server.dbnum; j++) {
          long long keys, vkeys;
  
@@ -5816,7 +6174,7 @@ static void sendBulkToSlave(aeEventLoop *el, int fd, void *privdata, int mask) {
          return;
      }
      if ((nwritten = write(fd,buf,buflen)) == -1) {
          return;
      }
      if ((nwritten = write(fd,buf,buflen)) == -1) {
-        redisLog(REDIS_DEBUG,"Write error sending DB to slave: %s",
+        redisLog(REDIS_VERBOSE,"Write error sending DB to slave: %s",
              strerror(errno));
          freeClient(slave);
          return;
              strerror(errno));
          freeClient(slave);
          return;
@@ -6020,6 +6378,18 @@ static void slaveofCommand(redisClient *c) {
  
  /* ============================ Maxmemory directive  ======================== */
  
  
  /* ============================ Maxmemory directive  ======================== */
  
+/* Free one object form the pre-allocated objects free list. This is useful
+ * under low mem conditions as by default we take 1 million free objects
+ * allocated. */
+static void freeOneObjectFromFreelist(void) {
+    robj *o;
+
+    listNode *head = listFirst(server.objfreelist);
+    o = listNodeValue(head);
+    listDelNode(server.objfreelist,head);
+    zfree(o);
+}
+
  /* This function gets called when 'maxmemory' is set on the config file to limit
   * the max memory used by the server, and we are out of memory.
   * This function will try to, in order:
  /* This function gets called when 'maxmemory' is set on the config file to limit
   * the max memory used by the server, and we are out of memory.
   * This function will try to, in order:
@@ -6034,12 +6404,7 @@ static void slaveofCommand(redisClient *c) {
  static void freeMemoryIfNeeded(void) {
      while (server.maxmemory && zmalloc_used_memory() > server.maxmemory) {
          if (listLength(server.objfreelist)) {
  static void freeMemoryIfNeeded(void) {
      while (server.maxmemory && zmalloc_used_memory() > server.maxmemory) {
          if (listLength(server.objfreelist)) {
-            robj *o;
-
-            listNode *head = listFirst(server.objfreelist);
-            o = listNodeValue(head);
-            listDelNode(server.objfreelist,head);
-            zfree(o);
+            freeOneObjectFromFreelist();
          } else {
              int j, k, freed = 0;
  
          } else {
              int j, k, freed = 0;
  
@@ -6190,6 +6555,7 @@ int loadAppendOnlyFile(char *filename) {
      struct redisClient *fakeClient;
      FILE *fp = fopen(filename,"r");
      struct redis_stat sb;
      struct redisClient *fakeClient;
      FILE *fp = fopen(filename,"r");
      struct redis_stat sb;
+    unsigned long long loadedkeys = 0;
  
      if (redis_fstat(fileno(fp),&sb) != -1 && sb.st_size == 0)
          return REDIS_ERR;
  
      if (redis_fstat(fileno(fp),&sb) != -1 && sb.st_size == 0)
          return REDIS_ERR;
@@ -6251,6 +6617,13 @@ int loadAppendOnlyFile(char *filename) {
          /* Clean up, ready for the next command */
          for (j = 0; j < argc; j++) decrRefCount(argv[j]);
          zfree(argv);
          /* Clean up, ready for the next command */
          for (j = 0; j < argc; j++) decrRefCount(argv[j]);
          zfree(argv);
+        /* Handle swapping while loading big datasets when VM is on */
+        loadedkeys++;
+        if (server.vm_enabled && (loadedkeys % 5000) == 0) {
+            while (zmalloc_used_memory() > server.vm_max_memory) {
+                if (vmSwapOneObjectBlocking() == REDIS_ERR) break;
+            }
+        }
      }
      fclose(fp);
      freeFakeClient(fakeClient);
      }
      fclose(fp);
      freeFakeClient(fakeClient);
@@ -6271,16 +6644,21 @@ fmterr:
  /* Write an object into a file in the bulk format $<count>\r\n<payload>\r\n */
  static int fwriteBulk(FILE *fp, robj *obj) {
      char buf[128];
  /* Write an object into a file in the bulk format $<count>\r\n<payload>\r\n */
  static int fwriteBulk(FILE *fp, robj *obj) {
      char buf[128];
-    obj = getDecodedObject(obj);
+    int decrrc = 0;
+
+    if (obj->storage == REDIS_VM_MEMORY && obj->encoding != REDIS_ENCODING_RAW){
+        obj = getDecodedObject(obj);
+        decrrc = 1;
+    }
      snprintf(buf,sizeof(buf),"$%ld\r\n",(long)sdslen(obj->ptr));
      if (fwrite(buf,strlen(buf),1,fp) == 0) goto err;
      if (sdslen(obj->ptr) && fwrite(obj->ptr,sdslen(obj->ptr),1,fp) == 0)
          goto err;
      if (fwrite("\r\n",2,1,fp) == 0) goto err;
      snprintf(buf,sizeof(buf),"$%ld\r\n",(long)sdslen(obj->ptr));
      if (fwrite(buf,strlen(buf),1,fp) == 0) goto err;
      if (sdslen(obj->ptr) && fwrite(obj->ptr,sdslen(obj->ptr),1,fp) == 0)
          goto err;
      if (fwrite("\r\n",2,1,fp) == 0) goto err;
-    decrRefCount(obj);
+    if (decrrc) decrRefCount(obj);
      return 1;
  err:
      return 1;
  err:
-    decrRefCount(obj);
+    if (decrrc) decrRefCount(obj);
      return 0;
  }
  
      return 0;
  }
  
@@ -6341,9 +6719,24 @@ static int rewriteAppendOnlyFile(char *filename) {
  
          /* Iterate this DB writing every entry */
          while((de = dictNext(di)) != NULL) {
  
          /* Iterate this DB writing every entry */
          while((de = dictNext(di)) != NULL) {
-            robj *key = dictGetEntryKey(de);
-            robj *o = dictGetEntryVal(de);
-            time_t expiretime = getExpire(db,key);
+            robj *key, *o;
+            time_t expiretime;
+            int swapped;
+
+            key = dictGetEntryKey(de);
+            /* If the value for this key is swapped, load a preview in memory.
+             * We use a "swapped" flag to remember if we need to free the
+             * value object instead to just increment the ref count anyway
+             * in order to avoid copy-on-write of pages if we are forked() */
+            if (!server.vm_enabled || key->storage == REDIS_VM_MEMORY ||
+                key->storage == REDIS_VM_SWAPPING) {
+                o = dictGetEntryVal(de);
+                swapped = 0;
+            } else {
+                o = vmPreviewObject(key);
+                swapped = 1;
+            }
+            expiretime = getExpire(db,key);
  
              /* Save the key and associated value */
              if (o->type == REDIS_STRING) {
  
              /* Save the key and associated value */
              if (o->type == REDIS_STRING) {
@@ -6411,6 +6804,7 @@ static int rewriteAppendOnlyFile(char *filename) {
                  if (fwriteBulk(fp,key) == 0) goto werr;
                  if (fwriteBulkLong(fp,expiretime) == 0) goto werr;
              }
                  if (fwriteBulk(fp,key) == 0) goto werr;
                  if (fwriteBulkLong(fp,expiretime) == 0) goto werr;
              }
+            if (swapped) decrRefCount(o);
          }
          dictReleaseIterator(di);
      }
          }
          dictReleaseIterator(di);
      }
@@ -6506,6 +6900,705 @@ static void aofRemoveTempFile(pid_t childpid) {
      unlink(tmpfile);
  }
  
      unlink(tmpfile);
  }
  
+/* Virtual Memory is composed mainly of two subsystems:
+ * - Blocking Virutal Memory
+ * - Threaded Virtual Memory I/O
+ * The two parts are not fully decoupled, but functions are split among two
+ * different sections of the source code (delimited by comments) in order to
+ * make more clear what functionality is about the blocking VM and what about
+ * the threaded (not blocking) VM.
+ *
+ * Redis VM design:
+ *
+ * Redis VM is a blocking VM (one that blocks reading swapped values from
+ * disk into memory when a value swapped out is needed in memory) that is made
+ * unblocking by trying to examine the command argument vector in order to
+ * load in background values that will likely be needed in order to exec
+ * the command. The command is executed only once all the relevant keys
+ * are loaded into memory.
+ *
+ * This basically is almost as simple of a blocking VM, but almost as parallel
+ * as a fully non-blocking VM.
+ */
+
+/* =================== Virtual Memory - Blocking Side  ====================== */
+static void vmInit(void) {
+    off_t totsize;
+    int pipefds[2];
+
+    server.vm_fp = fopen("/tmp/redisvm","w+b");
+    if (server.vm_fp == NULL) {
+        redisLog(REDIS_WARNING,"Impossible to open the swap file. Exiting.");
+        exit(1);
+    }
+    server.vm_fd = fileno(server.vm_fp);
+    server.vm_next_page = 0;
+    server.vm_near_pages = 0;
+    server.vm_stats_used_pages = 0;
+    server.vm_stats_swapped_objects = 0;
+    server.vm_stats_swapouts = 0;
+    server.vm_stats_swapins = 0;
+    totsize = server.vm_pages*server.vm_page_size;
+    redisLog(REDIS_NOTICE,"Allocating %lld bytes of swap file",totsize);
+    if (ftruncate(server.vm_fd,totsize) == -1) {
+        redisLog(REDIS_WARNING,"Can't ftruncate swap file: %s. Exiting.",
+            strerror(errno));
+        exit(1);
+    } else {
+        redisLog(REDIS_NOTICE,"Swap file allocated with success");
+    }
+    server.vm_bitmap = zmalloc((server.vm_pages+7)/8);
+    redisLog(REDIS_VERBOSE,"Allocated %lld bytes page table for %lld pages",
+        (long long) (server.vm_pages+7)/8, server.vm_pages);
+    memset(server.vm_bitmap,0,(server.vm_pages+7)/8);
+    /* Try to remove the swap file, so the OS will really delete it from the
+     * file system when Redis exists. */
+    unlink("/tmp/redisvm");
+
+    /* Initialize threaded I/O (used by Virtual Memory) */
+    server.io_newjobs = listCreate();
+    server.io_processing = listCreate();
+    server.io_processed = listCreate();
+    server.io_clients = listCreate();
+    pthread_mutex_init(&server.io_mutex,NULL);
+    server.io_active_threads = 0;
+    if (pipe(pipefds) == -1) {
+        redisLog(REDIS_WARNING,"Unable to intialized VM: pipe(2): %s. Exiting."
+            ,strerror(errno));
+        exit(1);
+    }
+    server.io_ready_pipe_read = pipefds[0];
+    server.io_ready_pipe_write = pipefds[1];
+    redisAssert(anetNonBlock(NULL,server.io_ready_pipe_read) != ANET_ERR);
+    /* Listen for events in the threaded I/O pipe */
+    if (aeCreateFileEvent(server.el, server.io_ready_pipe_read, AE_READABLE,
+        vmThreadedIOCompletedJob, NULL) == AE_ERR)
+        oom("creating file event");
+}
+
+/* Mark the page as used */
+static void vmMarkPageUsed(off_t page) {
+    off_t byte = page/8;
+    int bit = page&7;
+    server.vm_bitmap[byte] |= 1<<bit;
+    redisLog(REDIS_DEBUG,"Mark used: %lld (byte:%lld bit:%d)\n",
+        (long long)page, (long long)byte, bit);
+}
+
+/* Mark N contiguous pages as used, with 'page' being the first. */
+static void vmMarkPagesUsed(off_t page, off_t count) {
+    off_t j;
+
+    for (j = 0; j < count; j++)
+        vmMarkPageUsed(page+j);
+    server.vm_stats_used_pages += count;
+}
+
+/* Mark the page as free */
+static void vmMarkPageFree(off_t page) {
+    off_t byte = page/8;
+    int bit = page&7;
+    server.vm_bitmap[byte] &= ~(1<<bit);
+}
+
+/* Mark N contiguous pages as free, with 'page' being the first. */
+static void vmMarkPagesFree(off_t page, off_t count) {
+    off_t j;
+
+    for (j = 0; j < count; j++)
+        vmMarkPageFree(page+j);
+    server.vm_stats_used_pages -= count;
+}
+
+/* Test if the page is free */
+static int vmFreePage(off_t page) {
+    off_t byte = page/8;
+    int bit = page&7;
+    return (server.vm_bitmap[byte] & (1<<bit)) == 0;
+}
+
+/* Find N contiguous free pages storing the first page of the cluster in *first.
+ * Returns REDIS_OK if it was able to find N contiguous pages, otherwise 
+ * REDIS_ERR is returned.
+ *
+ * This function uses a simple algorithm: we try to allocate
+ * REDIS_VM_MAX_NEAR_PAGES sequentially, when we reach this limit we start
+ * again from the start of the swap file searching for free spaces.
+ *
+ * If it looks pretty clear that there are no free pages near our offset
+ * we try to find less populated places doing a forward jump of
+ * REDIS_VM_MAX_RANDOM_JUMP, then we start scanning again a few pages
+ * without hurry, and then we jump again and so forth...
+ * 
+ * This function can be improved using a free list to avoid to guess
+ * too much, since we could collect data about freed pages.
+ *
+ * note: I implemented this function just after watching an episode of
+ * Battlestar Galactica, where the hybrid was continuing to say "JUMP!"
+ */
+static int vmFindContiguousPages(off_t *first, int n) {
+    off_t base, offset = 0, since_jump = 0, numfree = 0;
+
+    if (server.vm_near_pages == REDIS_VM_MAX_NEAR_PAGES) {
+        server.vm_near_pages = 0;
+        server.vm_next_page = 0;
+    }
+    server.vm_near_pages++; /* Yet another try for pages near to the old ones */
+    base = server.vm_next_page;
+
+    while(offset < server.vm_pages) {
+        off_t this = base+offset;
+
+        redisLog(REDIS_DEBUG, "THIS: %lld (%c)\n", (long long) this, vmFreePage(this) ? 'F' : 'X');
+        /* If we overflow, restart from page zero */
+        if (this >= server.vm_pages) {
+            this -= server.vm_pages;
+            if (this == 0) {
+                /* Just overflowed, what we found on tail is no longer
+                 * interesting, as it's no longer contiguous. */
+                numfree = 0;
+            }
+        }
+        if (vmFreePage(this)) {
+            /* This is a free page */
+            numfree++;
+            /* Already got N free pages? Return to the caller, with success */
+            if (numfree == n) {
+                *first = this-(n-1);
+                server.vm_next_page = this+1;
+                return REDIS_OK;
+            }
+        } else {
+            /* The current one is not a free page */
+            numfree = 0;
+        }
+
+        /* Fast-forward if the current page is not free and we already
+         * searched enough near this place. */
+        since_jump++;
+        if (!numfree && since_jump >= REDIS_VM_MAX_RANDOM_JUMP/4) {
+            offset += random() % REDIS_VM_MAX_RANDOM_JUMP;
+            since_jump = 0;
+            /* Note that even if we rewind after the jump, we are don't need
+             * to make sure numfree is set to zero as we only jump *if* it
+             * is set to zero. */
+        } else {
+            /* Otherwise just check the next page */
+            offset++;
+        }
+    }
+    return REDIS_ERR;
+}
+
+/* Swap the 'val' object relative to 'key' into disk. Store all the information
+ * needed to later retrieve the object into the key object.
+ * If we can't find enough contiguous empty pages to swap the object on disk
+ * REDIS_ERR is returned. */
+static int vmSwapObjectBlocking(robj *key, robj *val) {
+    off_t pages = rdbSavedObjectPages(val,NULL);
+    off_t page;
+
+    assert(key->storage == REDIS_VM_MEMORY);
+    assert(key->refcount == 1);
+    if (vmFindContiguousPages(&page,pages) == REDIS_ERR) return REDIS_ERR;
+    if (fseeko(server.vm_fp,page*server.vm_page_size,SEEK_SET) == -1) {
+        redisLog(REDIS_WARNING,
+            "Critical VM problem in vmSwapObjectBlocking(): can't seek: %s",
+            strerror(errno));
+        return REDIS_ERR;
+    }
+    rdbSaveObject(server.vm_fp,val);
+    key->vm.page = page;
+    key->vm.usedpages = pages;
+    key->storage = REDIS_VM_SWAPPED;
+    key->vtype = val->type;
+    decrRefCount(val); /* Deallocate the object from memory. */
+    vmMarkPagesUsed(page,pages);
+    redisLog(REDIS_DEBUG,"VM: object %s swapped out at %lld (%lld pages)",
+        (unsigned char*) key->ptr,
+        (unsigned long long) page, (unsigned long long) pages);
+    server.vm_stats_swapped_objects++;
+    server.vm_stats_swapouts++;
+    fflush(server.vm_fp);
+    return REDIS_OK;
+}
+
+/* Load the value object relative to the 'key' object from swap to memory.
+ * The newly allocated object is returned.
+ *
+ * If preview is true the unserialized object is returned to the caller but
+ * no changes are made to the key object, nor the pages are marked as freed */
+static robj *vmGenericLoadObject(robj *key, int preview) {
+    robj *val;
+
+    redisAssert(key->storage == REDIS_VM_SWAPPED);
+    if (fseeko(server.vm_fp,key->vm.page*server.vm_page_size,SEEK_SET) == -1) {
+        redisLog(REDIS_WARNING,
+            "Unrecoverable VM problem in vmLoadObject(): can't seek: %s",
+            strerror(errno));
+        exit(1);
+    }
+    val = rdbLoadObject(key->vtype,server.vm_fp);
+    if (val == NULL) {
+        redisLog(REDIS_WARNING, "Unrecoverable VM problem in vmLoadObject(): can't load object from swap file: %s", strerror(errno));
+        exit(1);
+    }
+    if (!preview) {
+        key->storage = REDIS_VM_MEMORY;
+        key->vm.atime = server.unixtime;
+        vmMarkPagesFree(key->vm.page,key->vm.usedpages);
+        redisLog(REDIS_DEBUG, "VM: object %s loaded from disk",
+            (unsigned char*) key->ptr);
+        server.vm_stats_swapped_objects--;
+    } else {
+        redisLog(REDIS_DEBUG, "VM: object %s previewed from disk",
+            (unsigned char*) key->ptr);
+    }
+    server.vm_stats_swapins++;
+    return val;
+}
+
+/* Plain object loading, from swap to memory */
+static robj *vmLoadObject(robj *key) {
+    /* If we are loading the object in background, stop it, we
+     * need to load this object synchronously ASAP. */
+    if (key->storage == REDIS_VM_LOADING)
+        vmCancelThreadedIOJob(key);
+    return vmGenericLoadObject(key,0);
+}
+
+/* Just load the value on disk, without to modify the key.
+ * This is useful when we want to perform some operation on the value
+ * without to really bring it from swap to memory, like while saving the
+ * dataset or rewriting the append only log. */
+static robj *vmPreviewObject(robj *key) {
+    return vmGenericLoadObject(key,1);
+}
+
+/* How a good candidate is this object for swapping?
+ * The better candidate it is, the greater the returned value.
+ *
+ * Currently we try to perform a fast estimation of the object size in
+ * memory, and combine it with aging informations.
+ *
+ * Basically swappability = idle-time * log(estimated size)
+ *
+ * Bigger objects are preferred over smaller objects, but not
+ * proportionally, this is why we use the logarithm. This algorithm is
+ * just a first try and will probably be tuned later. */
+static double computeObjectSwappability(robj *o) {
+    time_t age = server.unixtime - o->vm.atime;
+    long asize = 0;
+    list *l;
+    dict *d;
+    struct dictEntry *de;
+    int z;
+
+    if (age <= 0) return 0;
+    switch(o->type) {
+    case REDIS_STRING:
+        if (o->encoding != REDIS_ENCODING_RAW) {
+            asize = sizeof(*o);
+        } else {
+            asize = sdslen(o->ptr)+sizeof(*o)+sizeof(long)*2;
+        }
+        break;
+    case REDIS_LIST:
+        l = o->ptr;
+        listNode *ln = listFirst(l);
+
+        asize = sizeof(list);
+        if (ln) {
+            robj *ele = ln->value;
+            long elesize;
+
+            elesize = (ele->encoding == REDIS_ENCODING_RAW) ?
+                            (sizeof(*o)+sdslen(ele->ptr)) :
+                            sizeof(*o);
+            asize += (sizeof(listNode)+elesize)*listLength(l);
+        }
+        break;
+    case REDIS_SET:
+    case REDIS_ZSET:
+        z = (o->type == REDIS_ZSET);
+        d = z ? ((zset*)o->ptr)->dict : o->ptr;
+
+        asize = sizeof(dict)+(sizeof(struct dictEntry*)*dictSlots(d));
+        if (z) asize += sizeof(zset)-sizeof(dict);
+        if (dictSize(d)) {
+            long elesize;
+            robj *ele;
+
+            de = dictGetRandomKey(d);
+            ele = dictGetEntryKey(de);
+            elesize = (ele->encoding == REDIS_ENCODING_RAW) ?
+                            (sizeof(*o)+sdslen(ele->ptr)) :
+                            sizeof(*o);
+            asize += (sizeof(struct dictEntry)+elesize)*dictSize(d);
+            if (z) asize += sizeof(zskiplistNode)*dictSize(d);
+        }
+        break;
+    }
+    return (double)asize*log(1+asize);
+}
+
+/* Try to swap an object that's a good candidate for swapping.
+ * Returns REDIS_OK if the object was swapped, REDIS_ERR if it's not possible
+ * to swap any object at all.
+ *
+ * If 'usethreaded' is true, Redis will try to swap the object in background
+ * using I/O threads. */
+static int vmSwapOneObject(int usethreads) {
+    int j, i;
+    struct dictEntry *best = NULL;
+    double best_swappability = 0;
+    redisDb *best_db = NULL;
+    robj *key, *val;
+
+    for (j = 0; j < server.dbnum; j++) {
+        redisDb *db = server.db+j;
+        int maxtries = 1000;
+
+        if (dictSize(db->dict) == 0) continue;
+        for (i = 0; i < 5; i++) {
+            dictEntry *de;
+            double swappability;
+
+            if (maxtries) maxtries--;
+            de = dictGetRandomKey(db->dict);
+            key = dictGetEntryKey(de);
+            val = dictGetEntryVal(de);
+            if (key->storage != REDIS_VM_MEMORY) {
+                if (maxtries) i--; /* don't count this try */
+                continue;
+            }
+            swappability = computeObjectSwappability(val);
+            if (!best || swappability > best_swappability) {
+                best = de;
+                best_swappability = swappability;
+                best_db = db;
+            }
+        }
+    }
+    if (best == NULL) {
+        redisLog(REDIS_DEBUG,"No swappable key found!");
+        return REDIS_ERR;
+    }
+    key = dictGetEntryKey(best);
+    val = dictGetEntryVal(best);
+
+    redisLog(REDIS_DEBUG,"Key with best swappability: %s, %f",
+        key->ptr, best_swappability);
+
+    /* Unshare the key if needed */
+    if (key->refcount > 1) {
+        robj *newkey = dupStringObject(key);
+        decrRefCount(key);
+        key = dictGetEntryKey(best) = newkey;
+    }
+    /* Swap it */
+    if (usethreads) {
+        vmSwapObjectThreaded(key,val,best_db);
+        return REDIS_OK;
+    } else {
+        if (vmSwapObjectBlocking(key,val) == REDIS_OK) {
+            dictGetEntryVal(best) = NULL;
+            return REDIS_OK;
+        } else {
+            return REDIS_ERR;
+        }
+    }
+}
+
+static int vmSwapOneObjectBlocking() {
+    return vmSwapOneObject(0);
+}
+
+static int vmSwapOneObjectThreaded() {
+    return vmSwapOneObject(1);
+}
+
+/* Return true if it's safe to swap out objects in a given moment.
+ * Basically we don't want to swap objects out while there is a BGSAVE
+ * or a BGAEOREWRITE running in backgroud. */
+static int vmCanSwapOut(void) {
+    return (server.bgsavechildpid == -1 && server.bgrewritechildpid == -1);
+}
+
+/* Delete a key if swapped. Returns 1 if the key was found, was swapped
+ * and was deleted. Otherwise 0 is returned. */
+static int deleteIfSwapped(redisDb *db, robj *key) {
+    dictEntry *de;
+    robj *foundkey;
+
+    if ((de = dictFind(db->dict,key)) == NULL) return 0;
+    foundkey = dictGetEntryKey(de);
+    if (foundkey->storage == REDIS_VM_MEMORY) return 0;
+    deleteKey(db,key);
+    return 1;
+}
+
+/* =================== Virtual Memory - Threaded I/O  ======================= */
+
+static void freeIOJob(iojob *j) {
+    if (j->type == REDIS_IOJOB_PREPARE_SWAP ||
+        j->type == REDIS_IOJOB_DO_SWAP)
+        decrRefCount(j->val);
+    decrRefCount(j->key);
+    zfree(j);
+}
+
+/* Every time a thread finished a Job, it writes a byte into the write side
+ * of an unix pipe in order to "awake" the main thread, and this function
+ * is called. */
+static void vmThreadedIOCompletedJob(aeEventLoop *el, int fd, void *privdata,
+            int mask)
+{
+    char buf[1];
+    int retval;
+    REDIS_NOTUSED(el);
+    REDIS_NOTUSED(mask);
+    REDIS_NOTUSED(privdata);
+
+    /* For every byte we read in the read side of the pipe, there is one
+     * I/O job completed to process. */
+    while((retval = read(fd,buf,1)) == 1) {
+        iojob *j;
+        listNode *ln;
+        robj *key;
+        struct dictEntry *de;
+
+        redisLog(REDIS_DEBUG,"Processing I/O completed job");
+        assert(listLength(server.io_processed) != 0);
+
+        /* Get the processed element (the oldest one) */
+        lockThreadedIO();
+        ln = listFirst(server.io_processed);
+        j = ln->value;
+        listDelNode(server.io_processed,ln);
+        unlockThreadedIO();
+        /* If this job is marked as canceled, just ignore it */
+        if (j->canceled) {
+            freeIOJob(j);
+            continue;
+        }
+        /* Post process it in the main thread, as there are things we
+         * can do just here to avoid race conditions and/or invasive locks */
+        redisLog(REDIS_DEBUG,"Job type: %d, key at %p (%s) refcount: %d\n", j->type, (void*)j->key, (char*)j->key->ptr, j->key->refcount);
+        if (j->key->refcount <= 0) {
+            printf("Ooops ref count is <= 0!\n");
+            exit(1);
+        }
+        de = dictFind(j->db->dict,j->key);
+        assert(de != NULL);
+        key = dictGetEntryKey(de);
+        if (j->type == REDIS_IOJOB_LOAD) {
+            /* Key loaded, bring it at home */
+            key->storage = REDIS_VM_MEMORY;
+            key->vm.atime = server.unixtime;
+            vmMarkPagesFree(key->vm.page,key->vm.usedpages);
+            redisLog(REDIS_DEBUG, "VM: object %s loaded from disk (threaded)",
+                (unsigned char*) key->ptr);
+            server.vm_stats_swapped_objects--;
+            server.vm_stats_swapins++;
+            freeIOJob(j);
+        } else if (j->type == REDIS_IOJOB_PREPARE_SWAP) {
+            /* Now we know the amount of pages required to swap this object.
+             * Let's find some space for it, and queue this task again
+             * rebranded as REDIS_IOJOB_DO_SWAP. */
+            if (vmFindContiguousPages(&j->page,j->pages) == REDIS_ERR) {
+                /* Ooops... no space! */
+                freeIOJob(j);
+            } else {
+                j->type = REDIS_IOJOB_DO_SWAP;
+                lockThreadedIO();
+                queueIOJob(j);
+                unlockThreadedIO();
+            }
+        } else if (j->type == REDIS_IOJOB_DO_SWAP) {
+            robj *val;
+
+            /* Key swapped. We can finally free some memory. */
+            val = dictGetEntryVal(de);
+            key->vm.page = j->page;
+            key->vm.usedpages = j->pages;
+            key->storage = REDIS_VM_SWAPPED;
+            key->vtype = j->val->type;
+            decrRefCount(val); /* Deallocate the object from memory. */
+            dictGetEntryVal(de) = NULL;
+            vmMarkPagesUsed(j->page,j->pages);
+            redisLog(REDIS_DEBUG,
+                "VM: object %s swapped out at %lld (%lld pages) (threaded)",
+                (unsigned char*) key->ptr,
+                (unsigned long long) j->page, (unsigned long long) j->pages);
+            server.vm_stats_swapped_objects++;
+            server.vm_stats_swapouts++;
+            freeIOJob(j);
+            /* Put a few more swap requests in queue if we are still
+             * out of memory */
+            if (zmalloc_used_memory() > server.vm_max_memory) {
+                int more = 1;
+                while(more) {
+                    lockThreadedIO();
+                    more = listLength(server.io_newjobs) <
+                            (unsigned) server.vm_max_threads;
+                    unlockThreadedIO();
+                    /* Don't waste CPU time if swappable objects are rare. */
+                    if (vmSwapOneObjectThreaded() == REDIS_ERR) break;
+                }
+            }
+        }
+    }
+    if (retval < 0 && errno != EAGAIN) {
+        redisLog(REDIS_WARNING,
+            "WARNING: read(2) error in vmThreadedIOCompletedJob() %s",
+            strerror(errno));
+    }
+}
+
+static void lockThreadedIO(void) {
+    pthread_mutex_lock(&server.io_mutex);
+}
+
+static void unlockThreadedIO(void) {
+    pthread_mutex_unlock(&server.io_mutex);
+}
+
+/* Remove the specified object from the threaded I/O queue if still not
+ * processed, otherwise make sure to flag it as canceled. */
+static void vmCancelThreadedIOJob(robj *o) {
+    list *lists[3] = {
+        server.io_newjobs, server.io_processing, server.io_processed
+    };
+    int i;
+
+    assert(o->storage == REDIS_VM_LOADING || o->storage == REDIS_VM_SWAPPING);
+    lockThreadedIO();
+    /* Search for a matching key in one of the queues */
+    for (i = 0; i < 3; i++) {
+        listNode *ln;
+
+        listRewind(lists[i]);
+        while ((ln = listYield(lists[i])) != NULL) {
+            iojob *job = ln->value;
+
+            if (compareStringObjects(job->key,o) == 0) {
+                switch(i) {
+                case 0: /* io_newjobs */
+                    /* If the job was not yet processed the best thing to do
+                     * is to remove it from the queue at all */
+                    decrRefCount(job->key);
+                    if (job->type == REDIS_IOJOB_PREPARE_SWAP ||
+                        job->type == REDIS_IOJOB_DO_SWAP)
+                        decrRefCount(job->val);
+                    listDelNode(lists[i],ln);
+                    zfree(job);
+                    break;
+                case 1: /* io_processing */
+                case 2: /* io_processed */
+                    job->canceled = 1;
+                    break;
+                }
+                if (o->storage == REDIS_VM_LOADING)
+                    o->storage = REDIS_VM_SWAPPED;
+                else if (o->storage == REDIS_VM_SWAPPING)
+                    o->storage = REDIS_VM_MEMORY;
+                unlockThreadedIO();
+                return;
+            }
+        }
+    }
+    unlockThreadedIO();
+    assert(1 != 1); /* We should never reach this */
+}
+
+static void *IOThreadEntryPoint(void *arg) {
+    iojob *j;
+    listNode *ln;
+    REDIS_NOTUSED(arg);
+
+    pthread_detach(pthread_self());
+    while(1) {
+        /* Get a new job to process */
+        lockThreadedIO();
+        if (listLength(server.io_newjobs) == 0) {
+            /* No new jobs in queue, exit. */
+            printf("Thread %lld exiting, nothing to do\n",
+                (long long) pthread_self());
+            server.io_active_threads--;
+            unlockThreadedIO();
+            return NULL;
+        }
+        ln = listFirst(server.io_newjobs);
+        j = ln->value;
+        listDelNode(server.io_newjobs,ln);
+        /* Add the job in the processing queue */
+        j->thread = pthread_self();
+        listAddNodeTail(server.io_processing,j);
+        ln = listLast(server.io_processing); /* We use ln later to remove it */
+        unlockThreadedIO();
+        printf("Thread %lld got a new job: %p about key '%s'\n",
+            (long long) pthread_self(), (void*)j, (char*)j->key->ptr);
+
+        /* Process the Job */
+        if (j->type == REDIS_IOJOB_LOAD) {
+        } else if (j->type == REDIS_IOJOB_PREPARE_SWAP) {
+            FILE *fp = fopen("/dev/null","w+");
+            j->pages = rdbSavedObjectPages(j->val,fp);
+            fclose(fp);
+        } else if (j->type == REDIS_IOJOB_DO_SWAP) {
+        }
+
+        /* Done: insert the job into the processed queue */
+        printf("Thread %lld completed the job: %p\n",
+            (long long) pthread_self(), (void*)j);
+        lockThreadedIO();
+        listDelNode(server.io_processing,ln);
+        listAddNodeTail(server.io_processed,j);
+        unlockThreadedIO();
+        
+        /* Signal the main thread there is new stuff to process */
+        assert(write(server.io_ready_pipe_write,"x",1) == 1);
+    }
+    return NULL; /* never reached */
+}
+
+static void spawnIOThread(void) {
+    pthread_t thread;
+
+    pthread_create(&thread,NULL,IOThreadEntryPoint,NULL);
+    server.io_active_threads++;
+}
+
+/* This function must be called while with threaded IO locked */
+static void queueIOJob(iojob *j) {
+    listAddNodeTail(server.io_newjobs,j);
+    if (server.io_active_threads < server.vm_max_threads)
+        spawnIOThread();
+}
+
+static int vmSwapObjectThreaded(robj *key, robj *val, redisDb *db) {
+    iojob *j;
+    
+    assert(key->storage == REDIS_VM_MEMORY);
+    assert(key->refcount == 1);
+
+    j = zmalloc(sizeof(*j));
+    j->type = REDIS_IOJOB_PREPARE_SWAP;
+    j->db = db;
+    j->key = dupStringObject(key);
+    j->val = val;
+    incrRefCount(val);
+    j->canceled = 0;
+    j->thread = (pthread_t) -1;
+    key->storage = REDIS_VM_SWAPPING;
+
+    lockThreadedIO();
+    queueIOJob(j);
+    unlockThreadedIO();
+    return REDIS_OK;
+}
+
  /* ================================= Debugging ============================== */
  
  static void debugCommand(redisClient *c) {
  /* ================================= Debugging ============================== */
  
  static void debugCommand(redisClient *c) {
@@ -6541,13 +7634,52 @@ static void debugCommand(redisClient *c) {
          }
          key = dictGetEntryKey(de);
          val = dictGetEntryVal(de);
          }
          key = dictGetEntryKey(de);
          val = dictGetEntryVal(de);
-        addReplySds(c,sdscatprintf(sdsempty(),
-            "+Key at:%p refcount:%d, value at:%p refcount:%d encoding:%d\r\n",
+        if (server.vm_enabled && (key->storage == REDIS_VM_MEMORY ||
+                                  key->storage == REDIS_VM_SWAPPING)) {
+            addReplySds(c,sdscatprintf(sdsempty(),
+                "+Key at:%p refcount:%d, value at:%p refcount:%d "
+                "encoding:%d serializedlength:%lld\r\n",
                  (void*)key, key->refcount, (void*)val, val->refcount,
                  (void*)key, key->refcount, (void*)val, val->refcount,
-                val->encoding));
+                val->encoding, rdbSavedObjectLen(val,NULL)));
+        } else {
+            addReplySds(c,sdscatprintf(sdsempty(),
+                "+Key at:%p refcount:%d, value swapped at: page %llu "
+                "using %llu pages\r\n",
+                (void*)key, key->refcount, (unsigned long long) key->vm.page,
+                (unsigned long long) key->vm.usedpages));
+        }
+    } else if (!strcasecmp(c->argv[1]->ptr,"swapout") && c->argc == 3) {
+        dictEntry *de = dictFind(c->db->dict,c->argv[2]);
+        robj *key, *val;
+
+        if (!server.vm_enabled) {
+            addReplySds(c,sdsnew("-ERR Virtual Memory is disabled\r\n"));
+            return;
+        }
+        if (!de) {
+            addReply(c,shared.nokeyerr);
+            return;
+        }
+        key = dictGetEntryKey(de);
+        val = dictGetEntryVal(de);
+        /* If the key is shared we want to create a copy */
+        if (key->refcount > 1) {
+            robj *newkey = dupStringObject(key);
+            decrRefCount(key);
+            key = dictGetEntryKey(de) = newkey;
+        }
+        /* Swap it */
+        if (key->storage != REDIS_VM_MEMORY) {
+            addReplySds(c,sdsnew("-ERR This key is not in memory\r\n"));
+        } else if (vmSwapObjectBlocking(key,val) == REDIS_OK) {
+            dictGetEntryVal(de) = NULL;
+            addReply(c,shared.ok);
+        } else {
+            addReply(c,shared.err);
+        }
      } else {
          addReplySds(c,sdsnew(
      } else {
          addReplySds(c,sdsnew(
-            "-ERR Syntax error, try DEBUG [SEGFAULT|OBJECT <key>|RELOAD]\r\n"));
+            "-ERR Syntax error, try DEBUG [SEGFAULT|OBJECT <key>|SWAPOUT <key>|RELOAD]\r\n"));
      }
  }
  
      }
  }
  
@@ -6633,8 +7765,6 @@ int main(int argc, char **argv) {
          if (rdbLoad(server.dbfilename) == REDIS_OK)
              redisLog(REDIS_NOTICE,"DB loaded from disk");
      }
          if (rdbLoad(server.dbfilename) == REDIS_OK)
              redisLog(REDIS_NOTICE,"DB loaded from disk");
      }
-    if (aeCreateFileEvent(server.el, server.fd, AE_READABLE,
-        acceptHandler, NULL) == AE_ERR) oom("creating file event");
      redisLog(REDIS_NOTICE,"The server is now ready to accept connections on port %d", server.port);
      aeMain(server.el);
      aeDeleteEventLoop(server.el);
      redisLog(REDIS_NOTICE,"The server is now ready to accept connections on port %d", server.port);
      aeMain(server.el);
      aeDeleteEventLoop(server.el);