mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
* BUG: Bad page state in process with linux 3.4.76
@ 2014-01-14 20:34 Guillaume Morin
  2014-01-14 22:10 ` Guillaume Morin
  0 siblings, 1 reply; 4+ messages in thread
From: Guillaume Morin @ 2014-01-14 20:34 UTC (permalink / raw)
  To: linux-kernel

[-- Attachment #1: Type: text/plain, Size: 2275 bytes --]

Hi,

I wrote this simple program (attached) to play around with kernel AIO.
It simply does kernel AIO with O_DIRECT on a small temp file stored on
an ext4 filesystem.

When I run it with "HUGETLB_MORECORE=yes LD_PRELOAD=libhugetlbfs.so", it
triggers the kernel bug on exit every time.

Removing HUGETLB_MORECORE from the command line fixes the problem.  Note
that my kernel does not use THP, it is NOT compiled with
CONFIG_TRANSPARENT_HUGEPAGE.

I've tried it only with this 3.4.76 but I've been able to reproduce it without
any issue on multiple machines running the same kernel.

BUG: Bad page state in process aio_test  pfn:1b7a01
page:ffffea0006de8040 count:0 mapcount:1 mapping:          (null) index:0x0
page flags: 0x20000000008000(tail)
Modules linked in: nfsd exportfs nfs nfs_acl auth_rpcgss fscache lockd sunrpc
rdma_ucm rdma_cm ib_addr iw_cm ib_uverbs ib_cm ib_sa ib_mad ib_core ipmi_si
ipmi_devintf coretemp pcspkr microcode serio_raw i2c_i801 ioatdma i2c_core dca
dm_mod sg sr_mod cdrom crc32c_intel ahci libahci [last unloaded: scsi_wait_scan]
Pid: 4441, comm: aio_test Not tainted 3.4.76bug #1
Call Trace:
[<ffffffff810f3300>] ? is_free_buddy_page+0xa0/0xd0
[<ffffffff814c0791>] bad_page+0xe6/0xfc
[<ffffffff810f3dbc>] free_pages_prepare+0xfc/0x110
[<ffffffff810f3dff>] __free_pages_ok+0x2f/0xd0
[<ffffffff810f4080>] __free_pages+0x20/0x40
[<ffffffff81124737>] update_and_free_page+0x77/0x80
[<ffffffff8112633e>] free_huge_page+0x16e/0x180
[<ffffffff810f8030>] __put_compound_page+0x20/0x50
[<ffffffff810f8108>] put_compound_page+0x78/0x140
[<ffffffff810f8546>] put_page+0x36/0x40
[<ffffffff81126ede>] __unmap_hugepage_range+0x1ce/0x230
[<ffffffff81127331>] unmap_hugepage_range+0x51/0x90
[<ffffffff8110e880>] unmap_single_vma+0x730/0x740
[<ffffffff8110f05f>] unmap_vmas+0x5f/0x80
[<ffffffff8111672c>] exit_mmap+0xbc/0x130
[<ffffffff8112e170>] ? kmem_cache_free+0x20/0xe0
[<ffffffff81035155>] mmput+0x35/0xf0
[<ffffffff8103a58d>] exit_mm+0xfd/0x120
[<ffffffff8103bb6c>] do_exit+0x16c/0x8b0
[<ffffffff811540c4>] ? mntput+0x24/0x40
[<ffffffff81138962>] ? fput+0x192/0x250
[<ffffffff8103c5ff>] do_group_exit+0x3f/0xa0
[<ffffffff8103c677>] sys_exit_group+0x17/0x20
[<ffffffff814d03d2>] system_call_fastpath+0x16/0x1b


-- 
Guillaume Morin <guillaume@morinfr.org>

[-- Attachment #2: aio_test.c --]
[-- Type: text/x-csrc, Size: 2434 bytes --]

#define _GNU_SOURCE
#include <libaio.h>
#include <errno.h>
#include <unistd.h>
#include <sys/types.h>
#include <sys/stat.h>
#include <fcntl.h>
#include <sys/eventfd.h>
#include <sys/epoll.h>
#include <sys/mman.h>
#include <stdio.h>
#include <stdlib.h>

#define FILE_SIZE 4096

int main(void)
{
    io_context_t ctx;
    int fd,fd_odirect,i,event_fd,epoll_fd;
    struct epoll_event ev;
    void *buf;
    size_t offset = 0;
    struct iocb cb;
    struct iocb * cbs[1] = { &cb };

    fd = open("/tmp/foo",O_RDWR|O_CREAT);
    if (fd == -1) {
        perror("open");
        return 1;
    }
    for (i = 0; i < FILE_SIZE; ++i) {
        char c = rand() % 255;
        write(fd, &c, 1);
    }
    close(fd);

    fd_odirect = open("/tmp/foo",O_RDONLY|O_DIRECT);
    if (fd_odirect == -1) {
        perror("open");
        return 1;
    }
    memset(&ctx, 0, sizeof(ctx));
    if (0 != io_queue_init(1, &ctx)) {
        perror("ctx");
        return 1;
    }
    event_fd = eventfd(0, EFD_CLOEXEC);
    if (event_fd == -1) {
        perror("eventfd");
        return -1;
    }

    epoll_fd = epoll_create(1);
    if (epoll_fd == -1) {
        perror("epoll_fd");
        return 1;
    }

    ev.events = EPOLLIN;

    if (epoll_ctl(epoll_fd, EPOLL_CTL_ADD, event_fd, &ev) == -1) {
        perror("epoll_ctl");
        return 1;
    }

    posix_memalign(&buf, 512, 32768);

    while (1) {
        struct timespec ts = { 0, 0 };
        struct io_event ioev;
        int ret;
        long v;
        io_prep_pread(&cb, fd_odirect, buf + offset, 512, offset);
        io_set_eventfd(&cb, event_fd);
        if (1 != io_submit(ctx, 1, cbs)) {
            perror("io_submit");
            return 1;
        }

        ret = epoll_wait(epoll_fd, &ev, 1, -1);
        if (ret != 1) {
            perror("epoll_wait");
        }

        read(event_fd, &v, 8);
        
        printf("event_fd returned %ld\n", v);


        if (io_getevents(ctx, 1, 1, &ioev, &ts) != 1) {
            perror("io_getevents");
            return 1;
        }

        printf("Read 1 res %ld res2 %ld\n", ioev.res, ioev.res2);
        offset += ioev.res;

        if (ioev.res == 0) {
            break;
        } 
        if ((offset + 512) > 32768) {
            puts("ERROR - reading past buffer");
            return 1;
        }
    }

    free(buf);
    io_destroy(ctx);
    close(event_fd);
    close(epoll_fd);
    close(fd_odirect);

    return 0;
}

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: BUG: Bad page state in process with linux 3.4.76
  2014-01-14 20:34 BUG: Bad page state in process with linux 3.4.76 Guillaume Morin
@ 2014-01-14 22:10 ` Guillaume Morin
  2014-02-24 11:39   ` Jan Kara
  0 siblings, 1 reply; 4+ messages in thread
From: Guillaume Morin @ 2014-01-14 22:10 UTC (permalink / raw)
  To: linux-kernel, stable

Greg,

I am going to do more testing but it seems that reverting this patch
from 3.4.69 fixes the BUG
commit b07ef016454ff46f98e633b5a6247ca7e343fb67
Author: Khalid Aziz <khalid.aziz@oracle.com>

I also verified that I cannot reproduce this problem with 3.13-rc8

Guillaume.

On 14 Jan 21:34, Guillaume Morin wrote:
>
> Hi,
> 
> I wrote this simple program (attached) to play around with kernel AIO.
> It simply does kernel AIO with O_DIRECT on a small temp file stored on
> an ext4 filesystem.
> 
> When I run it with "HUGETLB_MORECORE=yes LD_PRELOAD=libhugetlbfs.so", it
> triggers the kernel bug on exit every time.
> 
> Removing HUGETLB_MORECORE from the command line fixes the problem.  Note
> that my kernel does not use THP, it is NOT compiled with
> CONFIG_TRANSPARENT_HUGEPAGE.
> 
> I've tried it only with this 3.4.76 but I've been able to reproduce it without
> any issue on multiple machines running the same kernel.
> 
> BUG: Bad page state in process aio_test  pfn:1b7a01
> page:ffffea0006de8040 count:0 mapcount:1 mapping:          (null) index:0x0
> page flags: 0x20000000008000(tail)
> Modules linked in: nfsd exportfs nfs nfs_acl auth_rpcgss fscache lockd sunrpc
> rdma_ucm rdma_cm ib_addr iw_cm ib_uverbs ib_cm ib_sa ib_mad ib_core ipmi_si
> ipmi_devintf coretemp pcspkr microcode serio_raw i2c_i801 ioatdma i2c_core dca
> dm_mod sg sr_mod cdrom crc32c_intel ahci libahci [last unloaded: scsi_wait_scan]
> Pid: 4441, comm: aio_test Not tainted 3.4.76bug #1
> Call Trace:
> [<ffffffff810f3300>] ? is_free_buddy_page+0xa0/0xd0
> [<ffffffff814c0791>] bad_page+0xe6/0xfc
> [<ffffffff810f3dbc>] free_pages_prepare+0xfc/0x110
> [<ffffffff810f3dff>] __free_pages_ok+0x2f/0xd0
> [<ffffffff810f4080>] __free_pages+0x20/0x40
> [<ffffffff81124737>] update_and_free_page+0x77/0x80
> [<ffffffff8112633e>] free_huge_page+0x16e/0x180
> [<ffffffff810f8030>] __put_compound_page+0x20/0x50
> [<ffffffff810f8108>] put_compound_page+0x78/0x140
> [<ffffffff810f8546>] put_page+0x36/0x40
> [<ffffffff81126ede>] __unmap_hugepage_range+0x1ce/0x230
> [<ffffffff81127331>] unmap_hugepage_range+0x51/0x90
> [<ffffffff8110e880>] unmap_single_vma+0x730/0x740
> [<ffffffff8110f05f>] unmap_vmas+0x5f/0x80
> [<ffffffff8111672c>] exit_mmap+0xbc/0x130
> [<ffffffff8112e170>] ? kmem_cache_free+0x20/0xe0
> [<ffffffff81035155>] mmput+0x35/0xf0
> [<ffffffff8103a58d>] exit_mm+0xfd/0x120
> [<ffffffff8103bb6c>] do_exit+0x16c/0x8b0
> [<ffffffff811540c4>] ? mntput+0x24/0x40
> [<ffffffff81138962>] ? fput+0x192/0x250
> [<ffffffff8103c5ff>] do_group_exit+0x3f/0xa0
> [<ffffffff8103c677>] sys_exit_group+0x17/0x20
> [<ffffffff814d03d2>] system_call_fastpath+0x16/0x1b
> 


-- 
Guillaume Morin <guillaume@morinfr.org>

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: BUG: Bad page state in process with linux 3.4.76
  2014-01-14 22:10 ` Guillaume Morin
@ 2014-02-24 11:39   ` Jan Kara
  2014-02-24 15:06     ` Guillaume Morin
  0 siblings, 1 reply; 4+ messages in thread
From: Jan Kara @ 2014-02-24 11:39 UTC (permalink / raw)
  To: Guillaume Morin; +Cc: linux-kernel, stable, Khalid Aziz

On Tue 14-01-14 23:10:40, Guillaume Morin wrote:
> Greg,
> 
> I am going to do more testing but it seems that reverting this patch
> from 3.4.69 fixes the BUG
> commit b07ef016454ff46f98e633b5a6247ca7e343fb67
> Author: Khalid Aziz <khalid.aziz@oracle.com>
> 
> I also verified that I cannot reproduce this problem with 3.13-rc8
  I'm going through some old emails... Did this get resolved with later 3.4
stable kernels? If not, I guess you should ping Greg / Khalid to either
revert that commit (I guess preferable given the nature of the change) or
merge some additional fixup...

								Honza

> On 14 Jan 21:34, Guillaume Morin wrote:
> >
> > Hi,
> > 
> > I wrote this simple program (attached) to play around with kernel AIO.
> > It simply does kernel AIO with O_DIRECT on a small temp file stored on
> > an ext4 filesystem.
> > 
> > When I run it with "HUGETLB_MORECORE=yes LD_PRELOAD=libhugetlbfs.so", it
> > triggers the kernel bug on exit every time.
> > 
> > Removing HUGETLB_MORECORE from the command line fixes the problem.  Note
> > that my kernel does not use THP, it is NOT compiled with
> > CONFIG_TRANSPARENT_HUGEPAGE.
> > 
> > I've tried it only with this 3.4.76 but I've been able to reproduce it without
> > any issue on multiple machines running the same kernel.
> > 
> > BUG: Bad page state in process aio_test  pfn:1b7a01
> > page:ffffea0006de8040 count:0 mapcount:1 mapping:          (null) index:0x0
> > page flags: 0x20000000008000(tail)
> > Modules linked in: nfsd exportfs nfs nfs_acl auth_rpcgss fscache lockd sunrpc
> > rdma_ucm rdma_cm ib_addr iw_cm ib_uverbs ib_cm ib_sa ib_mad ib_core ipmi_si
> > ipmi_devintf coretemp pcspkr microcode serio_raw i2c_i801 ioatdma i2c_core dca
> > dm_mod sg sr_mod cdrom crc32c_intel ahci libahci [last unloaded: scsi_wait_scan]
> > Pid: 4441, comm: aio_test Not tainted 3.4.76bug #1
> > Call Trace:
> > [<ffffffff810f3300>] ? is_free_buddy_page+0xa0/0xd0
> > [<ffffffff814c0791>] bad_page+0xe6/0xfc
> > [<ffffffff810f3dbc>] free_pages_prepare+0xfc/0x110
> > [<ffffffff810f3dff>] __free_pages_ok+0x2f/0xd0
> > [<ffffffff810f4080>] __free_pages+0x20/0x40
> > [<ffffffff81124737>] update_and_free_page+0x77/0x80
> > [<ffffffff8112633e>] free_huge_page+0x16e/0x180
> > [<ffffffff810f8030>] __put_compound_page+0x20/0x50
> > [<ffffffff810f8108>] put_compound_page+0x78/0x140
> > [<ffffffff810f8546>] put_page+0x36/0x40
> > [<ffffffff81126ede>] __unmap_hugepage_range+0x1ce/0x230
> > [<ffffffff81127331>] unmap_hugepage_range+0x51/0x90
> > [<ffffffff8110e880>] unmap_single_vma+0x730/0x740
> > [<ffffffff8110f05f>] unmap_vmas+0x5f/0x80
> > [<ffffffff8111672c>] exit_mmap+0xbc/0x130
> > [<ffffffff8112e170>] ? kmem_cache_free+0x20/0xe0
> > [<ffffffff81035155>] mmput+0x35/0xf0
> > [<ffffffff8103a58d>] exit_mm+0xfd/0x120
> > [<ffffffff8103bb6c>] do_exit+0x16c/0x8b0
> > [<ffffffff811540c4>] ? mntput+0x24/0x40
> > [<ffffffff81138962>] ? fput+0x192/0x250
> > [<ffffffff8103c5ff>] do_group_exit+0x3f/0xa0
> > [<ffffffff8103c677>] sys_exit_group+0x17/0x20
> > [<ffffffff814d03d2>] system_call_fastpath+0x16/0x1b
> > 
> 
> 
> -- 
> Guillaume Morin <guillaume@morinfr.org>
> --
> To unsubscribe from this list: send the line "unsubscribe linux-kernel" in
> the body of a message to majordomo@vger.kernel.org
> More majordomo info at  http://vger.kernel.org/majordomo-info.html
> Please read the FAQ at  http://www.tux.org/lkml/
-- 
Jan Kara <jack@suse.cz>
SUSE Labs, CR

^ permalink raw reply	[flat|nested] 4+ messages in thread

* Re: BUG: Bad page state in process with linux 3.4.76
  2014-02-24 11:39   ` Jan Kara
@ 2014-02-24 15:06     ` Guillaume Morin
  0 siblings, 0 replies; 4+ messages in thread
From: Guillaume Morin @ 2014-02-24 15:06 UTC (permalink / raw)
  To: Jan Kara; +Cc: Guillaume Morin, linux-kernel, stable, Khalid Aziz

On 24 Feb 12:39, Jan Kara wrote:
>   I'm going through some old emails... Did this get resolved with later 3.4
> stable kernels? If not, I guess you should ping Greg / Khalid to either
> revert that commit (I guess preferable given the nature of the change) or
> merge some additional fixup...

Yes, it did get resolved.  Khalid backported
27c73ae759774e63313c1fbfeb17ba076cea64c5 which fixed the problem in the
mainline kernel and it was released in 3.4.79 as
50d8f1b5c57bb29f02ab5834be334b4f7922b856 (and included the other stable
branches as well).

Guillaume.

-- 
Guillaume Morin <guillaume@morinfr.org>

^ permalink raw reply	[flat|nested] 4+ messages in thread

end of thread, other threads:[~2014-02-24 15:06 UTC | newest]

Thread overview: 4+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2014-01-14 20:34 BUG: Bad page state in process with linux 3.4.76 Guillaume Morin
2014-01-14 22:10 ` Guillaume Morin
2014-02-24 11:39   ` Jan Kara
2014-02-24 15:06     ` Guillaume Morin

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®