Groups | Search | Server Info | Keyboard shortcuts | Login | Register [http] [https] [nntp] [nntps]


Groups > linux.kernel > #1252780 > unrolled thread

[PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

Started byTetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
First post2015-10-21 14:30 +0200
Last post2015-10-22 17:40 +0200
Articles 20 on this page of 52 — 5 participants

Back to article view | Back to linux.kernel


Contents

  [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-21 14:30 +0200
    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-21 15:10 +0200
    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-21 16:30 +0200
      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-21 16:40 +0200
        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-21 16:50 +0200
          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-21 17:00 +0200
            Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-21 17:40 +0200
            Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-21 19:20 +0200
              Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-22 13:40 +0200
                Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 15:40 +0200
                  Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 16:20 +0200
                    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 16:30 +0200
                      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 16:30 +0200
                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 16:30 +0200
                          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 16:40 +0200
                            Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 16:50 +0200
                              Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 17:20 +0200
                                Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-23 06:30 +0200
                      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 16:30 +0200
                    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 16:30 +0200
                    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-22 17:10 +0200
                      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 17:20 +0200
                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Christoph Lameter <cl@linux.com> - 2015-10-22 17:40 +0200
                          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-23 10:40 +0200
                            Make vmstat deferrable again (was Re: [PATCH] mm,vmscan: Use accurate  values for zone_reclaimable() checks) Christoph Lameter <cl@linux.com> - 2015-10-23 13:50 +0200
                              Re: Make vmstat deferrable again (was Re: [PATCH] mm,vmscan: Use  accurate values for zone_reclaimable() checks) Sergey Senozhatsky <sergey.senozhatsky@gmail.com> - 2015-10-23 14:10 +0200
                                Re: Make vmstat deferrable again (was Re: [PATCH] mm,vmscan: Use  accurate values for zone_reclaimable() checks) Christoph Lameter <cl@linux.com> - 2015-10-23 16:20 +0200
                                  Re: Make vmstat deferrable again (was Re: [PATCH] mm,vmscan: Use  accurate values for zone_reclaimable() checks) Sergey Senozhatsky <sergey.senozhatsky@gmail.com> - 2015-10-23 17:00 +0200
                                    Re: Make vmstat deferrable again (was Re: [PATCH] mm,vmscan: Use  accurate values for zone_reclaimable() checks) Christoph Lameter <cl@linux.com> - 2015-10-23 18:20 +0200
                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 17:40 +0200
                          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-22 17:50 +0200
                            Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-22 20:50 +0200
                              Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-22 23:50 +0200
                                Re: [PATCH] mm,vmscan: Use accurate values for  zone_reclaimable()checks Tejun Heo <htejun@gmail.com> - 2015-10-23 00:50 +0200
                                Re: [PATCH] mm,vmscan: Use accurate values for  zone_reclaimable()checks Michal Hocko <mhocko@kernel.org> - 2015-10-23 10:40 +0200
                                  Re: [PATCH] mm,vmscan: Use accurate values for  zone_reclaimable()checks Tejun Heo <htejun@gmail.com> - 2015-10-23 12:40 +0200
                              Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-23 10:40 +0200
                                Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-23 12:40 +0200
                                  Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-23 13:20 +0200
                                    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-23 14:30 +0200
                                      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-23 20:30 +0200
                                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-25 12:00 +0100
                                          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-25 23:50 +0100
                                          Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-27 10:30 +0100
                                            Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-27 12:00 +0100
                                              Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-27 13:10 +0100
                                    Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-23 20:30 +0200
                                      Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-27 10:20 +0100
                                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Tejun Heo <htejun@gmail.com> - 2015-10-27 12:00 +0100
                                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()checks Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp> - 2015-10-27 12:10 +0100
                                          Re: [PATCH] mm,vmscan: Use accurate values for  zone_reclaimable()checks Tejun Heo <htejun@gmail.com> - 2015-10-27 12:40 +0100
                        Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable()  checks Michal Hocko <mhocko@kernel.org> - 2015-10-22 17:40 +0200

Page 1 of 3  [1] 2 3  Next page →


#1252780 — [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
Date2015-10-21 14:30 +0200
Subject[PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm0GK-2RY-1@gated-at.bofh.it>
>From 0c50792dfa6396453c89c71351a7458b94d3e881 Mon Sep 17 00:00:00 2001
From: Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
Date: Wed, 21 Oct 2015 21:15:30 +0900
Subject: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

Since "struct zone"->vm_stat[] is array of atomic_long_t, an attempt
to reduce frequency of updating values in vm_stat[] is achieved by
using per cpu variables "struct per_cpu_pageset"->vm_stat_diff[].
Values in vm_stat_diff[] are merged into vm_stat[] periodically
(configured via /proc/sys/vm/stat_interval) using vmstat_update
workqueue (struct delayed_work vmstat_work).

When a task attempted to allocate memory and reached direct reclaim
path, shrink_zones() checks whether there are reclaimable pages by
calling zone_reclaimable(). zone_reclaimable() makes decision based
on values in vm_stat[] by calling zone_page_state(). This is usually
fine because values in vm_stat_diff[] are expected to be merged into
vm_stat[] shortly.

However, if a workqueue which is processed before vmstat_update
workqueue is processed got stuck inside memory allocation request,
values in vm_stat_diff[] cannot be merged into vm_stat[]. As a result,
zone_reclaimable() continues using outdated vm_stat[] values and the
task which is doing direct reclaim path thinks that there are reclaimable
pages and therefore continues looping. The consequence is a silent
livelock (hang up without any kernel messages) because the OOM killer
will not be invoked.

We can hit such livelock by e.g. disk_events_workfn workqueue doing
memory allocation from bio_copy_kern().

[  255.054205] kworker/3:1     R  running task        0    45      2 0x00000008
[  255.056063] Workqueue: events_freezable_power_ disk_events_workfn
[  255.057715]  ffff88007f805680 ffff88007c55f6d0 ffffffff8116463d ffff88007c55f758
[  255.059705]  ffff88007f82b870 ffff88007c55f6e0 ffffffff811646be ffff88007c55f710
[  255.061694]  ffffffff811bdaf0 ffff88007f82b870 0000000000000400 0000000000000000
[  255.063690] Call Trace:
[  255.064664]  [<ffffffff8116463d>] ? __list_lru_count_one.isra.4+0x1d/0x80
[  255.066428]  [<ffffffff811646be>] ? list_lru_count_one+0x1e/0x20
[  255.068063]  [<ffffffff811bdaf0>] ? super_cache_count+0x50/0xd0
[  255.069666]  [<ffffffff8114ecf6>] ? shrink_slab.part.38+0xf6/0x2a0
[  255.071313]  [<ffffffff81151f78>] ? shrink_zone+0x2c8/0x2e0
[  255.072845]  [<ffffffff81152316>] ? do_try_to_free_pages+0x156/0x6d0
[  255.074527]  [<ffffffff810bc6b6>] ? mark_held_locks+0x66/0x90
[  255.076085]  [<ffffffff816ca797>] ? _raw_spin_unlock_irq+0x27/0x40
[  255.077727]  [<ffffffff810bc7d9>] ? trace_hardirqs_on_caller+0xf9/0x1c0
[  255.079451]  [<ffffffff81152924>] ? try_to_free_pages+0x94/0xc0
[  255.081045]  [<ffffffff81145b4a>] ? __alloc_pages_nodemask+0x72a/0xdb0
[  255.082761]  [<ffffffff8118cd06>] ? alloc_pages_current+0x96/0x1b0
[  255.084407]  [<ffffffff8133985d>] ? bio_alloc_bioset+0x20d/0x2d0
[  255.086032]  [<ffffffff8133aba4>] ? bio_copy_kern+0xc4/0x180
[  255.087584]  [<ffffffff81344f20>] ? blk_rq_map_kern+0x70/0x130
[  255.089161]  [<ffffffff814a334d>] ? scsi_execute+0x12d/0x160
[  255.090696]  [<ffffffff814a3474>] ? scsi_execute_req_flags+0x84/0xf0
[  255.092466]  [<ffffffff814b55f2>] ? sr_check_events+0xb2/0x2a0
[  255.094042]  [<ffffffff814c3223>] ? cdrom_check_events+0x13/0x30
[  255.095634]  [<ffffffff814b5a35>] ? sr_block_check_events+0x25/0x30
[  255.097278]  [<ffffffff813501fb>] ? disk_check_events+0x5b/0x150
[  255.098865]  [<ffffffff81350307>] ? disk_events_workfn+0x17/0x20
[  255.100451]  [<ffffffff810890b5>] ? process_one_work+0x1a5/0x420
[  255.102046]  [<ffffffff81089051>] ? process_one_work+0x141/0x420
[  255.103625]  [<ffffffff8108944b>] ? worker_thread+0x11b/0x490
[  255.105159]  [<ffffffff816c4e95>] ? __schedule+0x315/0xac0
[  255.106643]  [<ffffffff81089330>] ? process_one_work+0x420/0x420
[  255.108217]  [<ffffffff8108f4e9>] ? kthread+0xf9/0x110
[  255.109634]  [<ffffffff8108f3f0>] ? kthread_create_on_node+0x230/0x230
[  255.111307]  [<ffffffff816cb35f>] ? ret_from_fork+0x3f/0x70
[  255.112785]  [<ffffffff8108f3f0>] ? kthread_create_on_node+0x230/0x230

[  273.930846] Showing busy workqueues and worker pools:
[  273.932299] workqueue events: flags=0x0
[  273.933465]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=4/256
[  273.935120]     pending: vmpressure_work_fn, vmstat_shepherd, vmstat_update, vmw_fb_dirty_flush [vmwgfx]
[  273.937489] workqueue events_freezable: flags=0x4
[  273.938795]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
[  273.940446]     pending: vmballoon_work [vmw_balloon]
[  273.941973] workqueue events_power_efficient: flags=0x80
[  273.943491]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
[  273.945167]     pending: check_lifetime
[  273.946422] workqueue events_freezable_power_: flags=0x84
[  273.947890]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
[  273.949579]     in-flight: 45:disk_events_workfn
[  273.951103] workqueue ipv6_addrconf: flags=0x8
[  273.952447]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/1
[  273.954121]     pending: addrconf_verify_work
[  273.955541] workqueue xfs-reclaim/sda1: flags=0x4
[  273.957036]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
[  273.958847]     pending: xfs_reclaim_worker
[  273.960392] pool 6: cpus=3 node=0 flags=0x0 nice=0 workers=3 idle: 186 26

This patch changes zone_reclaimable() to use zone_page_state_snapshot()
in order to make sure that values in vm_stat_diff[] are taken into
account when making decision.

Signed-off-by: Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
---
 mm/vmscan.c | 10 +++++-----
 1 file changed, 5 insertions(+), 5 deletions(-)

diff --git a/mm/vmscan.c b/mm/vmscan.c
index af4f4c0..2e4ef60 100644
--- a/mm/vmscan.c
+++ b/mm/vmscan.c
@@ -196,19 +196,19 @@ static unsigned long zone_reclaimable_pages(struct zone *zone)
 {
 	unsigned long nr;
 
-	nr = zone_page_state(zone, NR_ACTIVE_FILE) +
-	     zone_page_state(zone, NR_INACTIVE_FILE);
+	nr = zone_page_state_snapshot(zone, NR_ACTIVE_FILE) +
+	     zone_page_state_snapshot(zone, NR_INACTIVE_FILE);
 
 	if (get_nr_swap_pages() > 0)
-		nr += zone_page_state(zone, NR_ACTIVE_ANON) +
-		      zone_page_state(zone, NR_INACTIVE_ANON);
+		nr += zone_page_state_snapshot(zone, NR_ACTIVE_ANON) +
+		      zone_page_state_snapshot(zone, NR_INACTIVE_ANON);
 
 	return nr;
 }
 
 bool zone_reclaimable(struct zone *zone)
 {
-	return zone_page_state(zone, NR_PAGES_SCANNED) <
+	return zone_page_state_snapshot(zone, NR_PAGES_SCANNED) <
 		zone_reclaimable_pages(zone) * 6;
 }
 
-- 
1.8.3.1
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [next] | [standalone]


#1252795 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromMichal Hocko <mhocko@kernel.org>
Date2015-10-21 15:10 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm1jr-3S5-5@gated-at.bofh.it>
In reply to#1252780
On Wed 21-10-15 21:26:19, Tetsuo Handa wrote:
> >From 0c50792dfa6396453c89c71351a7458b94d3e881 Mon Sep 17 00:00:00 2001
> From: Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
> Date: Wed, 21 Oct 2015 21:15:30 +0900
> Subject: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
> 
> Since "struct zone"->vm_stat[] is array of atomic_long_t, an attempt
> to reduce frequency of updating values in vm_stat[] is achieved by
> using per cpu variables "struct per_cpu_pageset"->vm_stat_diff[].
> Values in vm_stat_diff[] are merged into vm_stat[] periodically
> (configured via /proc/sys/vm/stat_interval) using vmstat_update
> workqueue (struct delayed_work vmstat_work).
> 
> When a task attempted to allocate memory and reached direct reclaim
> path, shrink_zones() checks whether there are reclaimable pages by
> calling zone_reclaimable(). zone_reclaimable() makes decision based
> on values in vm_stat[] by calling zone_page_state(). This is usually
> fine because values in vm_stat_diff[] are expected to be merged into
> vm_stat[] shortly.
> 
> However, if a workqueue which is processed before vmstat_update
> workqueue is processed got stuck inside memory allocation request,
> values in vm_stat_diff[] cannot be merged into vm_stat[]. As a result,
> zone_reclaimable() continues using outdated vm_stat[] values and the
> task which is doing direct reclaim path thinks that there are reclaimable
> pages and therefore continues looping. The consequence is a silent
> livelock (hang up without any kernel messages) because the OOM killer
> will not be invoked.
> 
> We can hit such livelock by e.g. disk_events_workfn workqueue doing
> memory allocation from bio_copy_kern().
> 
> [  255.054205] kworker/3:1     R  running task        0    45      2 0x00000008
> [  255.056063] Workqueue: events_freezable_power_ disk_events_workfn
> [  255.057715]  ffff88007f805680 ffff88007c55f6d0 ffffffff8116463d ffff88007c55f758
> [  255.059705]  ffff88007f82b870 ffff88007c55f6e0 ffffffff811646be ffff88007c55f710
> [  255.061694]  ffffffff811bdaf0 ffff88007f82b870 0000000000000400 0000000000000000
> [  255.063690] Call Trace:
> [  255.064664]  [<ffffffff8116463d>] ? __list_lru_count_one.isra.4+0x1d/0x80
> [  255.066428]  [<ffffffff811646be>] ? list_lru_count_one+0x1e/0x20
> [  255.068063]  [<ffffffff811bdaf0>] ? super_cache_count+0x50/0xd0
> [  255.069666]  [<ffffffff8114ecf6>] ? shrink_slab.part.38+0xf6/0x2a0
> [  255.071313]  [<ffffffff81151f78>] ? shrink_zone+0x2c8/0x2e0
> [  255.072845]  [<ffffffff81152316>] ? do_try_to_free_pages+0x156/0x6d0
> [  255.074527]  [<ffffffff810bc6b6>] ? mark_held_locks+0x66/0x90
> [  255.076085]  [<ffffffff816ca797>] ? _raw_spin_unlock_irq+0x27/0x40
> [  255.077727]  [<ffffffff810bc7d9>] ? trace_hardirqs_on_caller+0xf9/0x1c0
> [  255.079451]  [<ffffffff81152924>] ? try_to_free_pages+0x94/0xc0
> [  255.081045]  [<ffffffff81145b4a>] ? __alloc_pages_nodemask+0x72a/0xdb0
> [  255.082761]  [<ffffffff8118cd06>] ? alloc_pages_current+0x96/0x1b0
> [  255.084407]  [<ffffffff8133985d>] ? bio_alloc_bioset+0x20d/0x2d0
> [  255.086032]  [<ffffffff8133aba4>] ? bio_copy_kern+0xc4/0x180
> [  255.087584]  [<ffffffff81344f20>] ? blk_rq_map_kern+0x70/0x130
> [  255.089161]  [<ffffffff814a334d>] ? scsi_execute+0x12d/0x160
> [  255.090696]  [<ffffffff814a3474>] ? scsi_execute_req_flags+0x84/0xf0
> [  255.092466]  [<ffffffff814b55f2>] ? sr_check_events+0xb2/0x2a0
> [  255.094042]  [<ffffffff814c3223>] ? cdrom_check_events+0x13/0x30
> [  255.095634]  [<ffffffff814b5a35>] ? sr_block_check_events+0x25/0x30
> [  255.097278]  [<ffffffff813501fb>] ? disk_check_events+0x5b/0x150
> [  255.098865]  [<ffffffff81350307>] ? disk_events_workfn+0x17/0x20
> [  255.100451]  [<ffffffff810890b5>] ? process_one_work+0x1a5/0x420
> [  255.102046]  [<ffffffff81089051>] ? process_one_work+0x141/0x420
> [  255.103625]  [<ffffffff8108944b>] ? worker_thread+0x11b/0x490
> [  255.105159]  [<ffffffff816c4e95>] ? __schedule+0x315/0xac0
> [  255.106643]  [<ffffffff81089330>] ? process_one_work+0x420/0x420
> [  255.108217]  [<ffffffff8108f4e9>] ? kthread+0xf9/0x110
> [  255.109634]  [<ffffffff8108f3f0>] ? kthread_create_on_node+0x230/0x230
> [  255.111307]  [<ffffffff816cb35f>] ? ret_from_fork+0x3f/0x70
> [  255.112785]  [<ffffffff8108f3f0>] ? kthread_create_on_node+0x230/0x230
> 
> [  273.930846] Showing busy workqueues and worker pools:
> [  273.932299] workqueue events: flags=0x0
> [  273.933465]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=4/256
> [  273.935120]     pending: vmpressure_work_fn, vmstat_shepherd, vmstat_update, vmw_fb_dirty_flush [vmwgfx]
> [  273.937489] workqueue events_freezable: flags=0x4
> [  273.938795]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
> [  273.940446]     pending: vmballoon_work [vmw_balloon]
> [  273.941973] workqueue events_power_efficient: flags=0x80
> [  273.943491]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
> [  273.945167]     pending: check_lifetime
> [  273.946422] workqueue events_freezable_power_: flags=0x84
> [  273.947890]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
> [  273.949579]     in-flight: 45:disk_events_workfn
> [  273.951103] workqueue ipv6_addrconf: flags=0x8
> [  273.952447]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/1
> [  273.954121]     pending: addrconf_verify_work
> [  273.955541] workqueue xfs-reclaim/sda1: flags=0x4
> [  273.957036]   pwq 6: cpus=3 node=0 flags=0x0 nice=0 active=1/256
> [  273.958847]     pending: xfs_reclaim_worker
> [  273.960392] pool 6: cpus=3 node=0 flags=0x0 nice=0 workers=3 idle: 186 26
> 
> This patch changes zone_reclaimable() to use zone_page_state_snapshot()
> in order to make sure that values in vm_stat_diff[] are taken into
> account when making decision.

Longerm we definitely want to get rid of zone_reclaimable for the OOM
detection. I hope I can post a proposal for this shortly but this is
simple enough and easy to backport to older kernels.

I would even consider it a stable candidate. It should go back in years.
Delayed vmstat updates go way back but there were other changes in the
area but it seems that at least since d1908362ae0b9 ("vmscan: check
all_unreclaimable in direct reclaim path") we are relying on
zone_reclaimable and vmstat was depending on WQ at the time already.

> Signed-off-by: Tetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>

Acked-by: Michal Hocko <mhocko@suse.com>

Thanks!

> ---
>  mm/vmscan.c | 10 +++++-----
>  1 file changed, 5 insertions(+), 5 deletions(-)
> 
> diff --git a/mm/vmscan.c b/mm/vmscan.c
> index af4f4c0..2e4ef60 100644
> --- a/mm/vmscan.c
> +++ b/mm/vmscan.c
> @@ -196,19 +196,19 @@ static unsigned long zone_reclaimable_pages(struct zone *zone)
>  {
>  	unsigned long nr;
>  
> -	nr = zone_page_state(zone, NR_ACTIVE_FILE) +
> -	     zone_page_state(zone, NR_INACTIVE_FILE);
> +	nr = zone_page_state_snapshot(zone, NR_ACTIVE_FILE) +
> +	     zone_page_state_snapshot(zone, NR_INACTIVE_FILE);
>  
>  	if (get_nr_swap_pages() > 0)
> -		nr += zone_page_state(zone, NR_ACTIVE_ANON) +
> -		      zone_page_state(zone, NR_INACTIVE_ANON);
> +		nr += zone_page_state_snapshot(zone, NR_ACTIVE_ANON) +
> +		      zone_page_state_snapshot(zone, NR_INACTIVE_ANON);
>  
>  	return nr;
>  }
>  
>  bool zone_reclaimable(struct zone *zone)
>  {
> -	return zone_page_state(zone, NR_PAGES_SCANNED) <
> +	return zone_page_state_snapshot(zone, NR_PAGES_SCANNED) <
>  		zone_reclaimable_pages(zone) * 6;
>  }
>  
> -- 
> 1.8.3.1
> --
> To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
> the body of a message to majordomo@vger.kernel.org
> More majordomo info at  http://vger.kernel.org/majordomo-info.html
> Please read the FAQ at  http://www.tux.org/lkml/

-- 
Michal Hocko
SUSE Labs
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1252864 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-21 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm2yU-5Ck-51@gated-at.bofh.it>
In reply to#1252780
On Wed, 21 Oct 2015, Tetsuo Handa wrote:

> However, if a workqueue which is processed before vmstat_update
> workqueue is processed got stuck inside memory allocation request,
> values in vm_stat_diff[] cannot be merged into vm_stat[]. As a result,
> zone_reclaimable() continues using outdated vm_stat[] values and the
> task which is doing direct reclaim path thinks that there are reclaimable
> pages and therefore continues looping. The consequence is a silent
> livelock (hang up without any kernel messages) because the OOM killer
> will not be invoked.

The diffs will be merged if they reach a certain threshold regardless. You
can decrease that threshhold. See calculate_pressure_threshhold().

Why is the merging not occurring if a process gets stuck? Workrequests are
not blocked by a process being stuck doing memory allocation or reclaim.

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1252870 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromMichal Hocko <mhocko@kernel.org>
Date2015-10-21 16:40 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm2Iy-5OI-15@gated-at.bofh.it>
In reply to#1252864
On Wed 21-10-15 09:22:40, Christoph Lameter wrote:
> On Wed, 21 Oct 2015, Tetsuo Handa wrote:
> 
> > However, if a workqueue which is processed before vmstat_update
> > workqueue is processed got stuck inside memory allocation request,
> > values in vm_stat_diff[] cannot be merged into vm_stat[]. As a result,
> > zone_reclaimable() continues using outdated vm_stat[] values and the
> > task which is doing direct reclaim path thinks that there are reclaimable
> > pages and therefore continues looping. The consequence is a silent
> > livelock (hang up without any kernel messages) because the OOM killer
> > will not be invoked.
> 
> The diffs will be merged if they reach a certain threshold regardless. You
> can decrease that threshhold. See calculate_pressure_threshhold().

The thing is that they will not reach the threshold. The LRUs in this
particular case are empty so there is nothing scanned so
NR_PAGES_SCANNED doesn't increase.

> Why is the merging not occurring if a process gets stuck? Workrequests are
> not blocked by a process being stuck doing memory allocation or reclaim.

Because all the WQ workers are stuck somewhere, maybe in the memory
allocation which cannot make any progress and the vmstat update work is
queued behind them.

At least this is my current understanding.
-- 
Michal Hocko
SUSE Labs
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1252889 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-21 16:50 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm2Se-606-31@gated-at.bofh.it>
In reply to#1252870
On Wed, 21 Oct 2015, Michal Hocko wrote:

> Because all the WQ workers are stuck somewhere, maybe in the memory
> allocation which cannot make any progress and the vmstat update work is
> queued behind them.
>
> At least this is my current understanding.

Eww. Maybe need a queue that does not do such evil things as memory
allocation?

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1252900 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromMichal Hocko <mhocko@kernel.org>
Date2015-10-21 17:00 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm31U-6dh-27@gated-at.bofh.it>
In reply to#1252889
On Wed 21-10-15 09:49:07, Christoph Lameter wrote:
> On Wed, 21 Oct 2015, Michal Hocko wrote:
> 
> > Because all the WQ workers are stuck somewhere, maybe in the memory
> > allocation which cannot make any progress and the vmstat update work is
> > queued behind them.
> >
> > At least this is my current understanding.
> 
> Eww. Maybe need a queue that does not do such evil things as memory
> allocation?

I am not sure how to achieve that. Requiring non-sleeping worker would
work out but do we have enough users to add such an API?

I would rather see vmstat using dedicated kernel thread(s) for this this
purpose. We have discussed that in the past but it hasn't led anywhere.

Anyway the workaround for this issue seems to be pretty trivial and
shouldn't affect users out of direct reclaim much so it sounds good
enough to me. Longterm we should really get rid of scan_reclaimable from
the direct reclaim altogether.

-- 
Michal Hocko
SUSE Labs
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1252960

FromTetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
Date2015-10-21 17:40 +0200
Message-ID<qm3EB-7ex-13@gated-at.bofh.it>
In reply to#1252900
Michal Hocko wrote:
> On Wed 21-10-15 09:49:07, Christoph Lameter wrote:
> > On Wed, 21 Oct 2015, Michal Hocko wrote:
> > 
> > > Because all the WQ workers are stuck somewhere, maybe in the memory
> > > allocation which cannot make any progress and the vmstat update work is
> > > queued behind them.

After invoking the OOM killer, we can easily observe that vmstat_update
cannot be processed due to memory allocation by disk_events_workfn stalls.
http://lkml.kernel.org/r/201509120019.BJI48986.OOSVMJtOLFQHFF@I-love.SAKURA.ne.jp

I worried that blocking forever from workqueue is an exclusive occupation of
workqueue. In fact, changing to GFP_ATOMIC avoids this problem.
http://lkml.kernel.org/r/201503012017.EAD00571.HOOJVOStMFLFQF@I-love.SAKURA.ne.jp

Now we realized that we are hitting this problem before invoking the OOM
killer. The situation is similar to the case after the OOM killer is
invoked; there are no reclaimable pages but vmstat_update cannot be
processed. We are caught by a small difference of vmstat counter values.

> > >
> > > At least this is my current understanding.
> > 
> > Eww. Maybe need a queue that does not do such evil things as memory
> > allocation?
> 
> I am not sure how to achieve that. Requiring non-sleeping worker would
> work out but do we have enough users to add such an API?

If a queue does not need to sleep, can't that queue be processed from
timer context (e.g. mod_timer()) ?
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253076 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-21 19:20 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qm5dn-1aW-9@gated-at.bofh.it>
In reply to#1252900
On Wed, 21 Oct 2015, Michal Hocko wrote:

> I am not sure how to achieve that. Requiring non-sleeping worker would
> work out but do we have enough users to add such an API?
>
> I would rather see vmstat using dedicated kernel thread(s) for this this
> purpose. We have discussed that in the past but it hasn't led anywhere.

How about this one? I really would like to have the vm statistics work as
designed and apparently they no longer work right with the existing
workqueue mechanism.


From: Christoph Lameter <cl@linux.com>
Subject: vmstat: Create our own workqueue

Seems that vmstat needs its own workqueue now since the general
workqueue mechanism has been *enhanced* which means that the
vmstat_updates cannot run reliably but are being blocked by
work requests doing memory allocation. Which causes vmstat
to be unable to keep the counters up to date.

Bad. Fix this by creating our own workqueue.

Signed-off-by: Christoph Lameter <cl@linux.com>

Index: linux/mm/vmstat.c
===================================================================
--- linux.orig/mm/vmstat.c
+++ linux/mm/vmstat.c
@@ -1357,6 +1357,8 @@ static const struct file_operations proc
 #endif /* CONFIG_PROC_FS */

 #ifdef CONFIG_SMP
+static struct workqueue_struct *vmstat_wq;
+
 static DEFINE_PER_CPU(struct delayed_work, vmstat_work);
 int sysctl_stat_interval __read_mostly = HZ;
 static cpumask_var_t cpu_stat_off;
@@ -1369,7 +1371,7 @@ static void vmstat_update(struct work_st
 		 * to occur in the future. Keep on running the
 		 * update worker thread.
 		 */
-		schedule_delayed_work_on(smp_processor_id(),
+		queue_delayed_work_on(smp_processor_id(), vmstat_wq,
 			this_cpu_ptr(&vmstat_work),
 			round_jiffies_relative(sysctl_stat_interval));
 	} else {
@@ -1438,7 +1440,7 @@ static void vmstat_shepherd(struct work_
 		if (need_update(cpu) &&
 			cpumask_test_and_clear_cpu(cpu, cpu_stat_off))

-			schedule_delayed_work_on(cpu,
+			queue_delayed_work_on(cpu, vmstat_wq,
 				&per_cpu(vmstat_work, cpu), 0);

 	put_online_cpus();
@@ -1534,6 +1536,7 @@ static int __init setup_vmstat(void)
 	proc_create("vmstat", S_IRUGO, NULL, &proc_vmstat_file_operations);
 	proc_create("zoneinfo", S_IRUGO, NULL, &proc_zoneinfo_file_operations);
 #endif
+	vmstat_wq = alloc_workqueue("vmstat", WQ_FREEZABLE|WQ_MEM_RECLAIM, 0);
 	return 0;
 }
 module_init(setup_vmstat)
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253725

FromTetsuo Handa <penguin-kernel@I-love.SAKURA.ne.jp>
Date2015-10-22 13:40 +0200
Message-ID<qmmnT-18n-5@gated-at.bofh.it>
In reply to#1253076
Christoph Lameter wrote:
> On Wed, 21 Oct 2015, Michal Hocko wrote:
> 
> > I am not sure how to achieve that. Requiring non-sleeping worker would
> > work out but do we have enough users to add such an API?
> >
> > I would rather see vmstat using dedicated kernel thread(s) for this this
> > purpose. We have discussed that in the past but it hasn't led anywhere.
> 
> How about this one? I really would like to have the vm statistics work as
> designed and apparently they no longer work right with the existing
> workqueue mechanism.

No, it won't help. Adding a dedicated workqueue for vmstat_update job
merely moves that job from "events" to "vmstat" workqueue. The "vmstat"
workqueue after all appears in the list of busy workqueues.

The problem is that all workqueues are assigned the same CPU (cpus=2 in
below example tested on a 4 CPUs VM) and therefore only one job is
in-flight state. All other jobs are waiting for the in-flight job to
complete in the pending list while the in-flight job is blocked at memory
allocation.

The problem would be that the "struct task_struct" to execute vmstat_update
job does not exist, and will not be able to create one on demand because we
are stuck at __GFP_WAIT allocation. Therefore adding a dedicated kernel
thread for vmstat_update job would work. But ...

------------------------------------------------------------
[  133.132322] Showing busy workqueues and worker pools:
[  133.133878] workqueue events: flags=0x0
[  133.135215]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=2/256
[  133.137076]     pending: vmpressure_work_fn, vmw_fb_dirty_flush [vmwgfx]
[  133.139075] workqueue events_freezable_power_: flags=0x84
[  133.140745]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=2/256
[  133.142638]     in-flight: 20:disk_events_workfn
[  133.144199]     pending: disk_events_workfn
[  133.145699] workqueue vmstat: flags=0xc
[  133.147055]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=1/256
[  133.148910]     pending: vmstat_update
[  133.150354] pool 4: cpus=2 node=0 flags=0x0 nice=0 workers=4 idle: 43 189 183
[  133.174523] DMA32 zone_reclaimable: reclaim:2(30186,30162,2) free:11163(25154,-20) min:11163 pages_scanned:0(30158,0) prio:12
[  133.177264] DMA32 zone_reclaimable: reclaim:2(30189,30165,2) free:11163(25157,-20) min:11163 pages_scanned:0(30161,0) prio:11
[  133.180139] DMA32 zone_reclaimable: reclaim:2(30191,30167,2) free:11163(25159,-20) min:11163 pages_scanned:0(30163,0) prio:10
[  133.182847] DMA32 zone_reclaimable: reclaim:2(30194,30170,2) free:11163(25162,-20) min:11163 pages_scanned:0(30166,0) prio:9
[  133.207048] DMA32 zone_reclaimable: reclaim:2(30219,30195,2) free:11163(25187,-20) min:11163 pages_scanned:0(30191,0) prio:8
[  133.209770] DMA32 zone_reclaimable: reclaim:2(30221,30197,2) free:11163(25189,-20) min:11163 pages_scanned:0(30193,0) prio:7
[  133.212470] DMA32 zone_reclaimable: reclaim:2(30224,30200,2) free:11163(25192,-20) min:11163 pages_scanned:0(30196,0) prio:6
[  133.215149] DMA32 zone_reclaimable: reclaim:2(30227,30203,2) free:11163(25195,-20) min:11163 pages_scanned:0(30199,0) prio:5
[  133.239013] DMA32 zone_reclaimable: reclaim:2(30251,30227,2) free:11163(25219,-20) min:11163 pages_scanned:0(30223,0) prio:4
[  133.241688] DMA32 zone_reclaimable: reclaim:2(30253,30229,2) free:11163(25221,-20) min:11163 pages_scanned:0(30225,0) prio:3
[  133.244332] DMA32 zone_reclaimable: reclaim:2(30256,30232,2) free:11163(25224,-20) min:11163 pages_scanned:0(30228,0) prio:2
[  133.246919] DMA32 zone_reclaimable: reclaim:2(30258,30234,2) free:11163(25226,-20) min:11163 pages_scanned:0(30230,0) prio:1
[  133.270967] DMA32 zone_reclaimable: reclaim:2(30283,30259,2) free:11163(25251,-20) min:11163 pages_scanned:0(30255,0) prio:0
[  133.273587] DMA32 zone_reclaimable: reclaim:2(30285,30261,2) free:11163(25253,-20) min:11163 pages_scanned:0(30257,0) prio:12
[  133.276224] DMA32 zone_reclaimable: reclaim:2(30287,30263,2) free:11163(25255,-20) min:11163 pages_scanned:0(30259,0) prio:11
[  133.278852] DMA32 zone_reclaimable: reclaim:2(30290,30266,2) free:11163(25258,-20) min:11163 pages_scanned:0(30262,0) prio:10
[  133.302964] DMA32 zone_reclaimable: reclaim:2(30315,30291,2) free:11163(25283,-20) min:11163 pages_scanned:0(30287,0) prio:9
[  133.305518] DMA32 zone_reclaimable: reclaim:2(30317,30293,2) free:11163(25285,-20) min:11163 pages_scanned:0(30289,0) prio:8
[  133.308095] DMA32 zone_reclaimable: reclaim:2(30319,30295,2) free:11163(25287,-20) min:11163 pages_scanned:0(30291,0) prio:7
[  133.310683] DMA32 zone_reclaimable: reclaim:2(30322,30298,2) free:11163(25290,-20) min:11163 pages_scanned:0(30294,0) prio:6
[  133.334904] DMA32 zone_reclaimable: reclaim:2(30347,30323,2) free:11163(25315,-20) min:11163 pages_scanned:0(30319,0) prio:5
[  133.337590] DMA32 zone_reclaimable: reclaim:2(30349,30325,2) free:11163(25317,-20) min:11163 pages_scanned:0(30321,0) prio:4
[  133.340147] DMA32 zone_reclaimable: reclaim:2(30351,30327,2) free:11163(25319,-20) min:11163 pages_scanned:0(30323,0) prio:3
[  133.343436] DMA32 zone_reclaimable: reclaim:2(30355,30331,2) free:11163(25323,-20) min:11163 pages_scanned:0(30327,0) prio:2
[  133.367531] DMA32 zone_reclaimable: reclaim:2(30379,30355,2) free:11163(25347,-20) min:11163 pages_scanned:0(30351,0) prio:1
[  133.370261] DMA32 zone_reclaimable: reclaim:2(30382,30358,2) free:11163(25350,-20) min:11163 pages_scanned:0(30354,0) prio:0
[  133.372786] did_some_progress=1 at line 3380
[  143.153205] MemAlloc-Info: 10 stalling task, 0 dying task, 0 victim task.
[  143.154981] MemAlloc: a.out(11052) gfp=0x24280ca order=0 delay=40104
[  143.156698] MemAlloc: abrt-watch-log(1708) gfp=0x242014a order=0 delay=39655
[  143.158527] MemAlloc: kworker/2:0(20) gfp=0x2400000 order=0 delay=39627
[  143.160247] MemAlloc: tuned(2076) gfp=0x242014a order=0 delay=39625
[  143.161920] MemAlloc: rngd(1703) gfp=0x242014a order=0 delay=38870
[  143.163618] MemAlloc: systemd-journal(471) gfp=0x242014a order=0 delay=38135
[  143.165435] MemAlloc: crond(1720) gfp=0x242014a order=0 delay=36330
[  143.167095] MemAlloc: vmtoolsd(1900) gfp=0x242014a order=0 delay=36136
[  143.168936] MemAlloc: irqbalance(1702) gfp=0x242014a order=0 delay=31584
[  143.170656] MemAlloc: nmbd(4791) gfp=0x242014a order=0 delay=30483
[  143.213896] Showing busy workqueues and worker pools:
[  143.215429] workqueue events: flags=0x0
[  143.216763]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=3/256
[  143.218542]     pending: vmpressure_work_fn, vmw_fb_dirty_flush [vmwgfx], console_callback
[  143.220785] workqueue events_freezable_power_: flags=0x84
[  143.222438]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=2/256
[  143.224239]     in-flight: 20:disk_events_workfn
[  143.225644]     pending: disk_events_workfn
[  143.227032] workqueue vmstat: flags=0xc
[  143.228303]   pwq 4: cpus=2 node=0 flags=0x0 nice=0 active=1/256
[  143.229996]     pending: vmstat_update
[  143.231396] pool 4: cpus=2 node=0 flags=0x0 nice=0 workers=4 idle: 43 189 183
[  144.023799] sysrq: SysRq : Kill All Tasks
------------------------------------------------------------

do we need to use a dedicated kernel thread for vmstat_update job?
It seems to me that refresh_cpu_vm_stats() will not sleep if we remove
cond_resched(). If vmstat_update job does not need to sleep, why can't
we do that job from timer interrupts? We have add_timer_on() which the
workqueue is also using.

Moreover, do we need to use atomic_long_t counters from the beginning?
Can't we do something like

 (1) Each CPU updates its own per CPU counter ("struct per_cpu_pageset"
     ->vm_stat_diff[] ?).
 (2) Only one thread periodically reads snapshot of all per CPU counters
     and adds diff values between the latest snapshot and previous snapshot
     to global counters ("struct zone"->vm_stat[] ?), and save the latest
     snapshot as previous snapshot.
 (3) Anyone can read global counters at any time.

which would not use atomic operations at all because only one task updates
global counters while each CPU continues using per CPU counters.



Linus Torvalds wrote (off-list due to mobile post):
> Side note: it would probably be interesting to see exactly *what*
> allocation ends up screwing up using just a regular workqueue. I bet
> there are lots of other workqueue users where timelineness can be a
> big deal - they continue to work, but perhaps they cause bad
> performance if there are allocators in other workqueues that end up
> delaying them.

We might want to favor kernel threads and dying threads over normal threads.
It helps reducing TIF_MEMDIE stalls if the dependency is limited to tasks
sharing the same memory.
http://lkml.kernel.org/r/201509102318.GHG18789.OHMSLFJOQFOtFV@I-love.SAKURA.ne.jp
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253817 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-22 15:40 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmog3-3Ro-19@gated-at.bofh.it>
In reply to#1253725
On Thu, 22 Oct 2015, Tetsuo Handa wrote:

> The problem would be that the "struct task_struct" to execute vmstat_update
> job does not exist, and will not be able to create one on demand because we
> are stuck at __GFP_WAIT allocation. Therefore adding a dedicated kernel
> thread for vmstat_update job would work. But ...

Yuck. Can someone please get this major screwup out of the work queue
subsystem? Tejun?

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253854 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-22 16:20 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmoSK-4U2-11@gated-at.bofh.it>
In reply to#1253817
On Thu, Oct 22, 2015 at 08:39:11AM -0500, Christoph Lameter wrote:
> On Thu, 22 Oct 2015, Tetsuo Handa wrote:
> 
> > The problem would be that the "struct task_struct" to execute vmstat_update
> > job does not exist, and will not be able to create one on demand because we
> > are stuck at __GFP_WAIT allocation. Therefore adding a dedicated kernel
> > thread for vmstat_update job would work. But ...
> 
> Yuck. Can someone please get this major screwup out of the work queue
> subsystem? Tejun?

Hmmm?  Just use a dedicated workqueue with WQ_MEM_RECLAIM.  If
concurrency management is a problem and there's something live-locking
for that work item (really?), WQ_CPU_INTENSIVE escapes it.  If this is
a common occurrence that it makes sense to give vmstat higher
priority, set WQ_HIGHPRI.

Thanks.

-- 
tejun
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253868 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-22 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmp2q-56v-25@gated-at.bofh.it>
In reply to#1253854
On Thu, Oct 22, 2015 at 11:09:44PM +0900, Tejun Heo wrote:
> On Thu, Oct 22, 2015 at 08:39:11AM -0500, Christoph Lameter wrote:
> > On Thu, 22 Oct 2015, Tetsuo Handa wrote:
> > 
> > > The problem would be that the "struct task_struct" to execute vmstat_update
> > > job does not exist, and will not be able to create one on demand because we
> > > are stuck at __GFP_WAIT allocation. Therefore adding a dedicated kernel
> > > thread for vmstat_update job would work. But ...
> > 
> > Yuck. Can someone please get this major screwup out of the work queue
> > subsystem? Tejun?
> 
> Hmmm?  Just use a dedicated workqueue with WQ_MEM_RECLAIM.  If
> concurrency management is a problem and there's something live-locking
> for that work item (really?), WQ_CPU_INTENSIVE escapes it.  If this is
> a common occurrence that it makes sense to give vmstat higher
> priority, set WQ_HIGHPRI.

Oooh, HIGHPRI + CPU_INTENSIVE immediate scheduling guarantee got lost
while converting HIGHPRI to a separate pool but guaranteeing immediate
scheduling for CPU_INTENSIVE is trivial.  If vmstat requires that,
please let me know.

Thanks.

-- 
tejun
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253869 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-22 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmp2q-56v-27@gated-at.bofh.it>
In reply to#1253868
On Thu, Oct 22, 2015 at 09:23:54AM -0500, Christoph Lameter wrote:
> I guess we need that otherwise vm statistics are not updated while worker
> threads are blocking on memory reclaim.

And the blocking one is just constantly running?

-- 
tejun
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253872 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-22 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmp2r-56v-43@gated-at.bofh.it>
In reply to#1253869
On Thu, 22 Oct 2015, Tejun Heo wrote:

> On Thu, Oct 22, 2015 at 09:23:54AM -0500, Christoph Lameter wrote:
> > I guess we need that otherwise vm statistics are not updated while worker
> > threads are blocking on memory reclaim.
>
> And the blocking one is just constantly running?

I was told that there is just one task struct so additional work queue
items cannot be processed while waiting?

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253879 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-22 16:40 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmpc6-5kr-11@gated-at.bofh.it>
In reply to#1253872
On Thu, Oct 22, 2015 at 09:25:49AM -0500, Christoph Lameter wrote:
> On Thu, 22 Oct 2015, Tejun Heo wrote:
> 
> > On Thu, Oct 22, 2015 at 09:23:54AM -0500, Christoph Lameter wrote:
> > > I guess we need that otherwise vm statistics are not updated while worker
> > > threads are blocking on memory reclaim.
> >
> > And the blocking one is just constantly running?
> 
> I was told that there is just one task struct so additional work queue
> items cannot be processed while waiting?

lol, no, what it tries to do is trying to keep the number of RUNNING
workers at minimum so that minimum number of workers can be used and
work items are executed back-to-back on the same workers.  The moment
a work item blocks, the next worker kicks in and starts executing the
next work item in line.

The only way to hang the execution for a work item w/ WQ_MEM_RECLAIM
is to create a cyclic dependency on another work item and keep that
work item busy wait.  Workqueue thinks that work item is making
progress as it's running and doesn't schedule the next one.

(I was misremembering here) HIGHPRI originally was implemented
head-queueing on the same pool followed by immediate execution, so
could get around cases where this could happen, but that got lost
while converting it to a separate pool.  I can introduce another flag
to bypass concurrency management if necessary (it's kinda trivial) but
busy-waiting cyclic dependency is a pretty unusual thing.

If this is actually a legit busy-waiting cyclic dependency, just let
me know.

Thanks.

-- 
tejun
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253889 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-22 16:50 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmplM-5wh-23@gated-at.bofh.it>
In reply to#1253879
On Thu, 22 Oct 2015, Tejun Heo wrote:

> The only way to hang the execution for a work item w/ WQ_MEM_RECLAIM
> is to create a cyclic dependency on another work item and keep that
> work item busy wait.  Workqueue thinks that work item is making
> progress as it's running and doesn't schedule the next one.
>
> (I was misremembering here) HIGHPRI originally was implemented
> head-queueing on the same pool followed by immediate execution, so
> could get around cases where this could happen, but that got lost
> while converting it to a separate pool.  I can introduce another flag
> to bypass concurrency management if necessary (it's kinda trivial) but
> busy-waiting cyclic dependency is a pretty unusual thing.
>
> If this is actually a legit busy-waiting cyclic dependency, just let
> me know.

There is no dependency of the vmstat updater on anything.
They can run anytime. If there is a dependency then its created by the
kworker subsystem itself.


--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253907 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-22 17:20 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmpOO-6iz-29@gated-at.bofh.it>
In reply to#1253889
Hello,

On Thu, Oct 22, 2015 at 09:41:11AM -0500, Christoph Lameter wrote:
> > If this is actually a legit busy-waiting cyclic dependency, just let
> > me know.
> 
> There is no dependency of the vmstat updater on anything.
> They can run anytime. If there is a dependency then its created by the
> kworker subsystem itself.

Sure, the other direction is from workqueue concurrency detection.  I
was asking whether a work item can busy-wait on vmstat_update work
item cuz that's what confuses workqueue.  Looking at the original
dump, the pool has two idle workers indicating that the workqueue
wasn't short of execution resources and it really looks like that work
item was live-locking the pool.  I'll go ahead and add WQ_IMMEDIATE.

Thanks.

-- 
tejun
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1254327 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromTejun Heo <htejun@gmail.com>
Date2015-10-23 06:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmC9j-7jL-1@gated-at.bofh.it>
In reply to#1253907
Hello,

So, something like the following.  Just compile tested but this is
essentially partial revert of 3270476a6c0c ("workqueue: reimplement
WQ_HIGHPRI using a separate worker_pool") - resurrecting the old
WQ_HIGHPRI implementation under WQ_IMMEDIATE, so we know this works.
If for some reason, it gets decided against simply adding one jiffy
sleep, please let me know.  I'll verify the operation and post a
proper patch.  That said, given that this prolly needs -stable
backport and vmstat is likely to be the only user (busy loops are
really rare in the kernel after all), I think the better approach
would be reinstating the short sleep.

Thanks.

---
 include/linux/workqueue.h |    7 ++---
 kernel/workqueue.c        |   63 +++++++++++++++++++++++++++++++++++++++++++---
 2 files changed, 63 insertions(+), 7 deletions(-)

--- a/include/linux/workqueue.h
+++ b/include/linux/workqueue.h
@@ -278,9 +278,10 @@ enum {
 	WQ_UNBOUND		= 1 << 1, /* not bound to any cpu */
 	WQ_FREEZABLE		= 1 << 2, /* freeze during suspend */
 	WQ_MEM_RECLAIM		= 1 << 3, /* may be used for memory reclaim */
-	WQ_HIGHPRI		= 1 << 4, /* high priority */
-	WQ_CPU_INTENSIVE	= 1 << 5, /* cpu intensive workqueue */
-	WQ_SYSFS		= 1 << 6, /* visible in sysfs, see wq_sysfs_register() */
+	WQ_IMMEDIATE		= 1 << 4, /* bypass concurrency management */
+	WQ_HIGHPRI		= 1 << 5, /* high priority */
+	WQ_CPU_INTENSIVE	= 1 << 6, /* cpu intensive workqueue */
+	WQ_SYSFS		= 1 << 7, /* visible in sysfs, see wq_sysfs_register() */
 
 	/*
 	 * Per-cpu workqueues are generally preferred because they tend to
--- a/kernel/workqueue.c
+++ b/kernel/workqueue.c
@@ -68,6 +68,7 @@ enum {
 	 * attach_mutex to avoid changing binding state while
 	 * worker_attach_to_pool() is in progress.
 	 */
+	POOL_IMMEDIATE_PENDING	= 1 << 0,	/* WQ_IMMEDIATE items on queue */
 	POOL_DISASSOCIATED	= 1 << 2,	/* cpu can't serve workers */
 
 	/* worker flags */
@@ -731,7 +732,8 @@ static bool work_is_canceling(struct wor
 
 static bool __need_more_worker(struct worker_pool *pool)
 {
-	return !atomic_read(&pool->nr_running);
+	return !atomic_read(&pool->nr_running) ||
+		(pool->flags & POOL_IMMEDIATE_PENDING);
 }
 
 /*
@@ -757,7 +759,8 @@ static bool may_start_working(struct wor
 static bool keep_working(struct worker_pool *pool)
 {
 	return !list_empty(&pool->worklist) &&
-		atomic_read(&pool->nr_running) <= 1;
+		(atomic_read(&pool->nr_running) <= 1 ||
+		 (pool->flags & POOL_IMMEDIATE_PENDING));
 }
 
 /* Do we need a new worker?  Called from manager. */
@@ -1021,6 +1024,42 @@ static void move_linked_works(struct wor
 }
 
 /**
+ * pwq_determine_ins_pos - find insertion position
+ * @pwq: pwq a work is being queued for
+ *
+ * A work for @pwq is about to be queued on @pwq->pool, determine insertion
+ * position for the work.  If @pwq is for IMMEDIATE wq, the work item is
+ * queued at the head of the queue but in FIFO order with respect to other
+ * IMMEDIATE work items; otherwise, at the end of the queue.  This function
+ * also sets POOL_IMMEDIATE_PENDING flag to hint @pool that there are
+ * IMMEDIATE works pending.
+ *
+ * CONTEXT:
+ * spin_lock_irq(gcwq->lock).
+ *
+ * RETURNS:
+ * Pointer to insertion position.
+ */
+static struct list_head *pwq_determine_ins_pos(struct pool_workqueue *pwq)
+{
+	struct worker_pool *pool = pwq->pool;
+	struct work_struct *twork;
+
+	if (likely(!(pwq->wq->flags & WQ_IMMEDIATE)))
+		return &pool->worklist;
+
+	list_for_each_entry(twork, &pool->worklist, entry) {
+		struct pool_workqueue *tpwq = get_work_pwq(twork);
+
+		if (!(tpwq->wq->flags & WQ_IMMEDIATE))
+			break;
+	}
+
+	pool->flags |= POOL_IMMEDIATE_PENDING;
+	return &twork->entry;
+}
+
+/**
  * get_pwq - get an extra reference on the specified pool_workqueue
  * @pwq: pool_workqueue to get
  *
@@ -1081,9 +1120,10 @@ static void put_pwq_unlocked(struct pool
 static void pwq_activate_delayed_work(struct work_struct *work)
 {
 	struct pool_workqueue *pwq = get_work_pwq(work);
+	struct list_head *pos = pwq_determine_ins_pos(pwq);
 
 	trace_workqueue_activate_work(work);
-	move_linked_works(work, &pwq->pool->worklist, NULL);
+	move_linked_works(work, pos, NULL);
 	__clear_bit(WORK_STRUCT_DELAYED_BIT, work_data_bits(work));
 	pwq->nr_active++;
 }
@@ -1384,7 +1424,7 @@ retry:
 	if (likely(pwq->nr_active < pwq->max_active)) {
 		trace_workqueue_activate_work(work);
 		pwq->nr_active++;
-		worklist = &pwq->pool->worklist;
+		worklist = pwq_determine_ins_pos(pwq);
 	} else {
 		work_flags |= WORK_STRUCT_DELAYED;
 		worklist = &pwq->delayed_works;
@@ -1996,6 +2036,21 @@ __acquires(&pool->lock)
 	list_del_init(&work->entry);
 
 	/*
+	 * If IMMEDIATE_PENDING, check the next work, and, if IMMEDIATE,
+	 * wake up another worker; otherwise, clear IMMEDIATE_PENDING.
+	 */
+	if (unlikely(pool->flags & POOL_IMMEDIATE_PENDING)) {
+		struct work_struct *nwork = list_first_entry(&pool->worklist,
+						struct work_struct, entry);
+
+		if (!list_empty(&pool->worklist) &&
+		    get_work_pwq(nwork)->wq->flags & WQ_IMMEDIATE)
+			wake_up_worker(pool);
+		else
+			pool->flags &= ~POOL_IMMEDIATE_PENDING;
+	}
+
+	/*
 	 * CPU intensive works don't participate in concurrency management.
 	 * They're the scheduler's responsibility.  This takes @worker out
 	 * of concurrency management and the next code block will chain
--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253871 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-22 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmp2q-56v-29@gated-at.bofh.it>
In reply to#1253868
On Thu, 22 Oct 2015, Tejun Heo wrote:

> > Hmmm?  Just use a dedicated workqueue with WQ_MEM_RECLAIM.  If
> > concurrency management is a problem and there's something live-locking
> > for that work item (really?), WQ_CPU_INTENSIVE escapes it.  If this is
> > a common occurrence that it makes sense to give vmstat higher
> > priority, set WQ_HIGHPRI.
>
> Oooh, HIGHPRI + CPU_INTENSIVE immediate scheduling guarantee got lost
> while converting HIGHPRI to a separate pool but guaranteeing immediate
> scheduling for CPU_INTENSIVE is trivial.  If vmstat requires that,
> please let me know.

I guess we need that otherwise vm statistics are not updated while worker
threads are blocking on memory reclaim.

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


#1253870 — Re: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks

FromChristoph Lameter <cl@linux.com>
Date2015-10-22 16:30 +0200
SubjectRe: [PATCH] mm,vmscan: Use accurate values for zone_reclaimable() checks
Message-ID<qmp2q-56v-31@gated-at.bofh.it>
In reply to#1253854
On Thu, 22 Oct 2015, Tejun Heo wrote:

> > Yuck. Can someone please get this major screwup out of the work queue
> > subsystem? Tejun?
>
> Hmmm?  Just use a dedicated workqueue with WQ_MEM_RECLAIM.  If
> concurrency management is a problem and there's something live-locking
> for that work item (really?), WQ_CPU_INTENSIVE escapes it.  If this is
> a common occurrence that it makes sense to give vmstat higher
> priority, set WQ_HIGHPRI.

I did. Check the thread. The result was that other tasks were still
blocking the thread. Ok I did not use HIGHPRI here is a newer version:


From: Christoph Lameter <cl@linux.com>
Subject: vmstat: Create our own workqueue V2

V1->V2:
   - Add a couple of workqueue flags that may fix things.

Seems that vmstat needs its own workqueue now since the general
workqueue mechanism has been *enhanced* which means that the
vmstat_updates cannot run reliably but are being blocked by
work requests doing memory allocation. Which causes vmstat
to be unable to keep the counters up to date.

Bad. Fix this by creating our own workqueue.

Signed-off-by: Christoph Lameter <cl@linux.com>

Index: linux/mm/vmstat.c
===================================================================
--- linux.orig/mm/vmstat.c
+++ linux/mm/vmstat.c
@@ -1382,6 +1382,8 @@ static const struct file_operations proc
 #endif /* CONFIG_PROC_FS */

 #ifdef CONFIG_SMP
+static struct workqueue_struct *vmstat_wq;
+
 static DEFINE_PER_CPU(struct delayed_work, vmstat_work);
 int sysctl_stat_interval __read_mostly = HZ;
 static cpumask_var_t cpu_stat_off;
@@ -1394,7 +1396,7 @@ static void vmstat_update(struct work_st
 		 * to occur in the future. Keep on running the
 		 * update worker thread.
 		 */
-		schedule_delayed_work_on(smp_processor_id(),
+		queue_delayed_work_on(smp_processor_id(), vmstat_wq,
 			this_cpu_ptr(&vmstat_work),
 			round_jiffies_relative(sysctl_stat_interval));
 	} else {
@@ -1463,7 +1465,7 @@ static void vmstat_shepherd(struct work_
 		if (need_update(cpu) &&
 			cpumask_test_and_clear_cpu(cpu, cpu_stat_off))

-			schedule_delayed_work_on(cpu,
+			queue_delayed_work_on(cpu, vmstat_wq,
 				&per_cpu(vmstat_work, cpu), 0);

 	put_online_cpus();
@@ -1552,6 +1554,12 @@ static int __init setup_vmstat(void)

 	start_shepherd_timer();
 	cpu_notifier_register_done();
+	vmstat_wq = alloc_workqueue("vmstat",
+		WQ_FREEZABLE|
+		WQ_SYSFS|
+		WQ_MEM_RECLAIM|
+		WQ_HIGHPRI|
+		WQ_CPU_INTENSIVE, 0);
 #endif
 #ifdef CONFIG_PROC_FS
 	proc_create("buddyinfo", S_IRUGO, NULL, &fragmentation_file_operations);

--
To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
the body of a message to majordomo@vger.kernel.org
More majordomo info at  http://vger.kernel.org/majordomo-info.html
Please read the FAQ at  http://www.tux.org/lkml/

[toc] | [prev] | [next] | [standalone]


Page 1 of 3  [1] 2 3  Next page →

Back to top | Article view | linux.kernel


csiph-web