memory hotplug: update zone pcp at memory online

[safe/jmp/linux-2.6] / mm / page_alloc.c
diff --git a/mm/page_alloc.c b/mm/page_alloc.c

index ad7cd1c..1a3a893 100644 (file)
--- a/mm/page_alloc.c
+++ b/mm/page_alloc.c
@@ -817,13 +817,15 @@ __rmqueue_fallback(struct zone *zone, int order, int start_migratetype)
                          * agressive about taking ownership of free pages
                          */
                         if (unlikely(current_order >= (pageblock_order >> 1)) ||
-                                       start_migratetype == MIGRATE_RECLAIMABLE) {
+                                       start_migratetype == MIGRATE_RECLAIMABLE ||
+                                       page_group_by_mobility_disabled) {
                                 unsigned long pages;
                                 pages = move_freepages_block(zone, page,
                                                                 start_migratetype);
  
                                 /* Claim the whole block if over half of it is free */
-                               if (pages >= (1 << (pageblock_order-1)))
+                               if (pages >= (1 << (pageblock_order-1)) ||
+                                               page_group_by_mobility_disabled)
                                         set_pageblock_migratetype(page,
                                                                 start_migratetype);
  
@@ -882,7 +884,7 @@ retry_reserve:
   */
  static int rmqueue_bulk(struct zone *zone, unsigned int order, 
                         unsigned long count, struct list_head *list,
-                       int migratetype)
+                       int migratetype, int cold)
  {
         int i;
         
@@ -901,7 +903,10 @@ static int rmqueue_bulk(struct zone *zone, unsigned int order,
                  * merge IO requests if the physical pages are ordered
                  * properly.
                  */
-               list_add(&page->lru, list);
+               if (likely(cold == 0))
+                       list_add(&page->lru, list);
+               else
+                       list_add_tail(&page->lru, list);
                 set_page_private(page, migratetype);
                 list = &page->lru;
         }
@@ -1119,7 +1124,8 @@ again:
                 local_irq_save(flags);
                 if (!pcp->count) {
                         pcp->count = rmqueue_bulk(zone, 0,
-                                       pcp->batch, &pcp->list, migratetype);
+                                       pcp->batch, &pcp->list,
+                                       migratetype, cold);
                         if (unlikely(!pcp->count))
                                 goto failed;
                 }
@@ -1138,7 +1144,8 @@ again:
                 /* Allocate more to the pcp list if necessary */
                 if (unlikely(&page->lru == &pcp->list)) {
                         pcp->count += rmqueue_bulk(zone, 0,
-                                       pcp->batch, &pcp->list, migratetype);
+                                       pcp->batch, &pcp->list,
+                                       migratetype, cold);
                         page = list_entry(pcp->list.next, struct page, lru);
                 }
  
@@ -1620,10 +1627,6 @@ __alloc_pages_direct_reclaim(gfp_t gfp_mask, unsigned int order,
  
         /* We now go into synchronous reclaim */
         cpuset_memory_pressure_bump();
-
-       /*
-        * The task's cpuset might have expanded its set of allowable nodes
-        */
         p->flags |= PF_MEMALLOC;
         lockdep_set_current_reclaim_state(gfp_mask);
         reclaim_state.reclaimed_slab = 0;
@@ -1666,7 +1669,7 @@ __alloc_pages_high_priority(gfp_t gfp_mask, unsigned int order,
                         preferred_zone, migratetype);
  
                 if (!page && gfp_mask & __GFP_NOFAIL)
-                       congestion_wait(WRITE, HZ/50);
+                       congestion_wait(BLK_RW_ASYNC, HZ/50);
         } while (!page && (gfp_mask & __GFP_NOFAIL));
  
         return page;
@@ -1740,8 +1743,10 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order,
          * be using allocators in order of preference for an area that is
          * too large.
          */
-       if (WARN_ON_ONCE(order >= MAX_ORDER))
+       if (order >= MAX_ORDER) {
+               WARN_ON_ONCE(!(gfp_mask & __GFP_NOWARN));
                 return NULL;
+       }
  
         /*
          * GFP_THISNODE (meaning __GFP_THISNODE, __GFP_NORETRY and
@@ -1789,6 +1794,10 @@ rebalance:
         if (p->flags & PF_MEMALLOC)
                 goto nopage;
  
+       /* Avoid allocations with no watermarks from looping endlessly */
+       if (test_thread_flag(TIF_MEMDIE) && !(gfp_mask & __GFP_NOFAIL))
+               goto nopage;
+
         /* Try direct reclaim and then allocating */
         page = __alloc_pages_direct_reclaim(gfp_mask, order,
                                         zonelist, high_zoneidx,
@@ -1831,7 +1840,7 @@ rebalance:
         pages_reclaimed += did_some_progress;
         if (should_alloc_retry(gfp_mask, order, pages_reclaimed)) {
                 /* Wait for some write requests to complete then retry */
-               congestion_wait(WRITE, HZ/50);
+               congestion_wait(BLK_RW_ASYNC, HZ/50);
                 goto rebalance;
         }
  
@@ -2533,7 +2542,6 @@ static void build_zonelists(pg_data_t *pgdat)
         prev_node = local_node;
         nodes_clear(used_mask);
  
-       memset(node_load, 0, sizeof(node_load));
         memset(node_order, 0, sizeof(node_order));
         j = 0;
  
@@ -2642,6 +2650,9 @@ static int __build_all_zonelists(void *dummy)
  {
         int nid;
  
+#ifdef CONFIG_NUMA
+       memset(node_load, 0, sizeof(node_load));
+#endif
         for_each_online_node(nid) {
                 pg_data_t *pgdat = NODE_DATA(nid);
  
@@ -3131,6 +3142,32 @@ int zone_wait_table_init(struct zone *zone, unsigned long zone_size_pages)
         return 0;
  }
  
+static int __zone_pcp_update(void *data)
+{
+       struct zone *zone = data;
+       int cpu;
+       unsigned long batch = zone_batchsize(zone), flags;
+
+       for (cpu = 0; cpu < NR_CPUS; cpu++) {
+               struct per_cpu_pageset *pset;
+               struct per_cpu_pages *pcp;
+
+               pset = zone_pcp(zone, cpu);
+               pcp = &pset->pcp;
+
+               local_irq_save(flags);
+               free_pages_bulk(zone, pcp->count, &pcp->list, 0);
+               setup_pageset(pset, batch);
+               local_irq_restore(flags);
+       }
+       return 0;
+}
+
+void zone_pcp_update(struct zone *zone)
+{
+       stop_machine(__zone_pcp_update, zone, NULL);
+}
+
  static __meminit void zone_pcp_init(struct zone *zone)
  {
         int cpu;
@@ -4745,8 +4782,10 @@ void *__init alloc_large_system_hash(const char *tablename,
                          * some pages at the end of hash table which
                          * alloc_pages_exact() automatically does
                          */
-                       if (get_order(size) < MAX_ORDER)
+                       if (get_order(size) < MAX_ORDER) {
                                 table = alloc_pages_exact(size, GFP_ATOMIC);
+                               kmemleak_alloc(table, size, 1, GFP_ATOMIC);
+                       }
                 }
         } while (!table && size > PAGE_SIZE && --log2qty);
  
@@ -4764,16 +4803,6 @@ void *__init alloc_large_system_hash(const char *tablename,
         if (_hash_mask)
                 *_hash_mask = (1 << log2qty) - 1;
  
-       /*
-        * If hashdist is set, the table allocation is done with __vmalloc()
-        * which invokes the kmemleak_alloc() callback. This function may also
-        * be called before the slab and kmemleak are initialised when
-        * kmemleak simply buffers the request to be executed later
-        * (GFP_ATOMIC flag ignored in this case).
-        */
-       if (!hashdist)
-               kmemleak_alloc(table, size, 1, GFP_ATOMIC);
-
         return table;
  }