Skip to content

Commit e3bde95

Browse files
dvlasenktorvalds
authored andcommitted
include/linux/unaligned: force inlining of byteswap operations
Sometimes gcc mysteriously doesn't inline very small functions we expect to be inlined. See https://gcc.gnu.org/bugzilla/show_bug.cgi?id=66122 With this .config: http://busybox.net/~vda/kernel_config_OPTIMIZE_INLINING_and_Os, the following functions get deinlined many times. Examples of disassembly: <get_unaligned_be16> (24 copies, 108 calls): 66 8b 07 mov (%rdi),%ax 55 push %rbp 48 89 e5 mov %rsp,%rbp 86 e0 xchg %ah,%al 5d pop %rbp c3 retq <get_unaligned_be32> (25 copies, 181 calls): 8b 07 mov (%rdi),%eax 55 push %rbp 48 89 e5 mov %rsp,%rbp 0f c8 bswap %eax 5d pop %rbp c3 retq <get_unaligned_be64> (23 copies, 94 calls): 48 8b 07 mov (%rdi),%rax 55 push %rbp 48 89 e5 mov %rsp,%rbp 48 0f c8 bswap %rax 5d pop %rbp c3 retq <put_unaligned_be16> (2 copies, 11 calls): 89 f8 mov %edi,%eax 55 push %rbp c1 ef 08 shr $0x8,%edi c1 e0 08 shl $0x8,%eax 09 c7 or %eax,%edi 48 89 e5 mov %rsp,%rbp 66 89 3e mov %di,(%rsi) <put_unaligned_be32> (8 copies, 43 calls): 55 push %rbp 0f cf bswap %edi 89 3e mov %edi,(%rsi) 48 89 e5 mov %rsp,%rbp 5d pop %rbp c3 retq <put_unaligned_be64> (26 copies, 157 calls): 55 push %rbp 48 0f cf bswap %rdi 48 89 3e mov %rdi,(%rsi) 48 89 e5 mov %rsp,%rbp 5d pop %rbp c3 retq This patch fixes this via s/inline/__always_inline/. It only affects arches with efficient unaligned access insns, such as x86. (arched which lack such ops do not include linux/unaligned/access_ok.h) Code size decrease after the patch is ~8.5k: text data bss dec hex filename 92197848 20826112 36417536 149441496 8e84bd8 vmlinux 92189231 20826144 36417536 149432911 8e82a4f vmlinux6_unaligned_be_after Signed-off-by: Denys Vlasenko <dvlasenk@redhat.com> Acked-by: Ingo Molnar <mingo@kernel.org> Cc: Thomas Graf <tgraf@suug.ch> Cc: Peter Zijlstra <peterz@infradead.org> Cc: David Rientjes <rientjes@google.com> Cc: Arnd Bergmann <arnd@arndb.de> Signed-off-by: Andrew Morton <akpm@linux-foundation.org> Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
1 parent bc27fb6 commit e3bde95

File tree

1 file changed

+12
-12
lines changed

1 file changed

+12
-12
lines changed

include/linux/unaligned/access_ok.h

Lines changed: 12 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -4,62 +4,62 @@
44
#include <linux/kernel.h>
55
#include <asm/byteorder.h>
66

7-
static inline u16 get_unaligned_le16(const void *p)
7+
static __always_inline u16 get_unaligned_le16(const void *p)
88
{
99
return le16_to_cpup((__le16 *)p);
1010
}
1111

12-
static inline u32 get_unaligned_le32(const void *p)
12+
static __always_inline u32 get_unaligned_le32(const void *p)
1313
{
1414
return le32_to_cpup((__le32 *)p);
1515
}
1616

17-
static inline u64 get_unaligned_le64(const void *p)
17+
static __always_inline u64 get_unaligned_le64(const void *p)
1818
{
1919
return le64_to_cpup((__le64 *)p);
2020
}
2121

22-
static inline u16 get_unaligned_be16(const void *p)
22+
static __always_inline u16 get_unaligned_be16(const void *p)
2323
{
2424
return be16_to_cpup((__be16 *)p);
2525
}
2626

27-
static inline u32 get_unaligned_be32(const void *p)
27+
static __always_inline u32 get_unaligned_be32(const void *p)
2828
{
2929
return be32_to_cpup((__be32 *)p);
3030
}
3131

32-
static inline u64 get_unaligned_be64(const void *p)
32+
static __always_inline u64 get_unaligned_be64(const void *p)
3333
{
3434
return be64_to_cpup((__be64 *)p);
3535
}
3636

37-
static inline void put_unaligned_le16(u16 val, void *p)
37+
static __always_inline void put_unaligned_le16(u16 val, void *p)
3838
{
3939
*((__le16 *)p) = cpu_to_le16(val);
4040
}
4141

42-
static inline void put_unaligned_le32(u32 val, void *p)
42+
static __always_inline void put_unaligned_le32(u32 val, void *p)
4343
{
4444
*((__le32 *)p) = cpu_to_le32(val);
4545
}
4646

47-
static inline void put_unaligned_le64(u64 val, void *p)
47+
static __always_inline void put_unaligned_le64(u64 val, void *p)
4848
{
4949
*((__le64 *)p) = cpu_to_le64(val);
5050
}
5151

52-
static inline void put_unaligned_be16(u16 val, void *p)
52+
static __always_inline void put_unaligned_be16(u16 val, void *p)
5353
{
5454
*((__be16 *)p) = cpu_to_be16(val);
5555
}
5656

57-
static inline void put_unaligned_be32(u32 val, void *p)
57+
static __always_inline void put_unaligned_be32(u32 val, void *p)
5858
{
5959
*((__be32 *)p) = cpu_to_be32(val);
6060
}
6161

62-
static inline void put_unaligned_be64(u64 val, void *p)
62+
static __always_inline void put_unaligned_be64(u64 val, void *p)
6363
{
6464
*((__be64 *)p) = cpu_to_be64(val);
6565
}

0 commit comments

Comments
 (0)