2005-04-17 02:20:36 +04:00
/ * Copyright 2 0 0 2 A n d i K l e e n , S u S E L a b s .
* Subject t o t h e G N U P u b l i c L i c e n s e v2 .
*
* Functions t o c o p y f r o m a n d t o u s e r s p a c e .
* /
2006-09-26 12:52:32 +04:00
# include < l i n u x / l i n k a g e . h >
# include < a s m / d w a r f2 . h >
2006-02-03 23:51:02 +03:00
# define F I X _ A L I G N M E N T 1
2006-09-26 12:52:39 +04:00
# include < a s m / c u r r e n t . h >
# include < a s m / a s m - o f f s e t s . h >
# include < a s m / t h r e a d _ i n f o . h >
# include < a s m / c p u f e a t u r e . h >
.macro ALTERNATIVE_JUMP feature,o r i g ,a l t
0 :
.byte 0xe9 /* 32bit jump */
.long \ orig- 1 f / * b y d e f a u l t j u m p t o o r i g * /
1 :
.section .altinstr_replacement , " ax"
2 : .byte 0xe9 /* near jump with 32bit immediate */
.long \ alt- 1 b / * o f f s e t * / / * o r a l t e r n a t i v e l y t o a l t * /
.previous
.section .altinstructions , " a"
.align 8
.quad 0b
.quad 2b
.byte \ feature / * w h e n f e a t u r e i s s e t * /
.byte 5
.byte 5
.previous
.endm
2005-04-17 02:20:36 +04:00
/* Standard copy_to_user with segment limit checking */
2006-09-26 12:52:32 +04:00
ENTRY( c o p y _ t o _ u s e r )
CFI_ S T A R T P R O C
2005-04-17 02:20:36 +04:00
GET_ T H R E A D _ I N F O ( % r a x )
movq % r d i ,% r c x
addq % r d x ,% r c x
jc b a d _ t o _ u s e r
cmpq t h r e a d i n f o _ a d d r _ l i m i t ( % r a x ) ,% r c x
jae b a d _ t o _ u s e r
2006-09-26 12:52:39 +04:00
xorl % e a x ,% e a x / * c l e a r z e r o f l a g * /
ALTERNATIVE_ J U M P X 8 6 _ F E A T U R E _ R E P _ G O O D ,c o p y _ u s e r _ g e n e r i c _ u n r o l l e d ,c o p y _ u s e r _ g e n e r i c _ s t r i n g
2006-09-26 12:52:32 +04:00
CFI_ E N D P R O C
2006-02-03 23:51:02 +03:00
2006-09-26 12:52:39 +04:00
ENTRY( c o p y _ u s e r _ g e n e r i c )
CFI_ S T A R T P R O C
movl $ 1 ,% e c x / * s e t z e r o f l a g * /
ALTERNATIVE_ J U M P X 8 6 _ F E A T U R E _ R E P _ G O O D ,c o p y _ u s e r _ g e n e r i c _ u n r o l l e d ,c o p y _ u s e r _ g e n e r i c _ s t r i n g
CFI_ E N D P R O C
ENTRY( _ _ c o p y _ f r o m _ u s e r _ i n a t o m i c )
CFI_ S T A R T P R O C
xorl % e c x ,% e c x / * c l e a r z e r o f l a g * /
ALTERNATIVE_ J U M P X 8 6 _ F E A T U R E _ R E P _ G O O D ,c o p y _ u s e r _ g e n e r i c _ u n r o l l e d ,c o p y _ u s e r _ g e n e r i c _ s t r i n g
CFI_ E N D P R O C
2005-04-17 02:20:36 +04:00
/* Standard copy_from_user with segment limit checking */
2006-09-26 12:52:32 +04:00
ENTRY( c o p y _ f r o m _ u s e r )
CFI_ S T A R T P R O C
2005-04-17 02:20:36 +04:00
GET_ T H R E A D _ I N F O ( % r a x )
movq % r s i ,% r c x
addq % r d x ,% r c x
jc b a d _ f r o m _ u s e r
cmpq t h r e a d i n f o _ a d d r _ l i m i t ( % r a x ) ,% r c x
jae b a d _ f r o m _ u s e r
2006-09-26 12:52:39 +04:00
movl $ 1 ,% e c x / * s e t z e r o f l a g * /
ALTERNATIVE_ J U M P X 8 6 _ F E A T U R E _ R E P _ G O O D ,c o p y _ u s e r _ g e n e r i c _ u n r o l l e d ,c o p y _ u s e r _ g e n e r i c _ s t r i n g
2006-09-26 12:52:32 +04:00
CFI_ E N D P R O C
ENDPROC( c o p y _ f r o m _ u s e r )
2005-04-17 02:20:36 +04:00
.section .fixup , " ax"
/* must zero dest */
bad_from_user :
2006-09-26 12:52:32 +04:00
CFI_ S T A R T P R O C
2005-04-17 02:20:36 +04:00
movl % e d x ,% e c x
xorl % e a x ,% e a x
rep
stosb
bad_to_user :
movl % e d x ,% e a x
ret
2006-09-26 12:52:32 +04:00
CFI_ E N D P R O C
END( b a d _ f r o m _ u s e r )
2005-04-17 02:20:36 +04:00
.previous
/ *
2006-09-26 12:52:39 +04:00
* copy_ u s e r _ g e n e r i c _ u n r o l l e d - m e m o r y c o p y w i t h e x c e p t i o n h a n d l i n g .
* This v e r s i o n i s f o r C P U s l i k e P 4 t h a t d o n ' t h a v e e f f i c i e n t m i c r o c o d e f o r r e p m o v s q
2005-04-17 02:20:36 +04:00
*
* Input :
* rdi d e s t i n a t i o n
* rsi s o u r c e
* rdx c o u n t
2006-09-26 12:52:39 +04:00
* ecx z e r o f l a g - - i f t r u e z e r o d e s t i n a t i o n o n e r r o r
2005-04-17 02:20:36 +04:00
*
* Output :
* eax u n c o p i e d b y t e s o r 0 i f s u c c e s s f u l .
* /
2006-09-26 12:52:39 +04:00
ENTRY( c o p y _ u s e r _ g e n e r i c _ u n r o l l e d )
2006-09-26 12:52:32 +04:00
CFI_ S T A R T P R O C
2006-02-03 23:51:02 +03:00
pushq % r b x
2006-09-26 12:52:32 +04:00
CFI_ A D J U S T _ C F A _ O F F S E T 8
CFI_ R E L _ O F F S E T r b x , 0
2006-09-26 12:52:39 +04:00
pushq % r c x
CFI_ A D J U S T _ C F A _ O F F S E T 8
CFI_ R E L _ O F F S E T r c x , 0
2006-02-03 23:51:02 +03:00
xorl % e a x ,% e a x / * z e r o f o r t h e e x c e p t i o n h a n d l e r * /
# ifdef F I X _ A L I G N M E N T
/* check for bad alignment of destination */
movl % e d i ,% e c x
andl $ 7 ,% e c x
jnz . L b a d _ a l i g n m e n t
.Lafter_bad_alignment :
# endif
movq % r d x ,% r c x
movl $ 6 4 ,% e b x
shrq $ 6 ,% r d x
decq % r d x
js . L h a n d l e _ t a i l
.p2align 4
.Lloop :
.Ls1 : movq ( % r s i ) ,% r11
.Ls2 : movq 1 * 8 ( % r s i ) ,% r8
.Ls3 : movq 2 * 8 ( % r s i ) ,% r9
.Ls4 : movq 3 * 8 ( % r s i ) ,% r10
.Ld1 : movq % r11 ,( % r d i )
.Ld2 : movq % r8 ,1 * 8 ( % r d i )
.Ld3 : movq % r9 ,2 * 8 ( % r d i )
.Ld4 : movq % r10 ,3 * 8 ( % r d i )
.Ls5 : movq 4 * 8 ( % r s i ) ,% r11
.Ls6 : movq 5 * 8 ( % r s i ) ,% r8
.Ls7 : movq 6 * 8 ( % r s i ) ,% r9
.Ls8 : movq 7 * 8 ( % r s i ) ,% r10
.Ld5 : movq % r11 ,4 * 8 ( % r d i )
.Ld6 : movq % r8 ,5 * 8 ( % r d i )
.Ld7 : movq % r9 ,6 * 8 ( % r d i )
.Ld8 : movq % r10 ,7 * 8 ( % r d i )
decq % r d x
leaq 6 4 ( % r s i ) ,% r s i
leaq 6 4 ( % r d i ) ,% r d i
jns . L l o o p
.p2align 4
.Lhandle_tail :
movl % e c x ,% e d x
andl $ 6 3 ,% e c x
shrl $ 3 ,% e c x
jz . L h a n d l e _ 7
movl $ 8 ,% e b x
.p2align 4
.Lloop_8 :
.Ls9 : movq ( % r s i ) ,% r8
.Ld9 : movq % r8 ,( % r d i )
decl % e c x
leaq 8 ( % r d i ) ,% r d i
leaq 8 ( % r s i ) ,% r s i
jnz . L l o o p _ 8
.Lhandle_7 :
movl % e d x ,% e c x
andl $ 7 ,% e c x
jz . L e n d e
.p2align 4
.Lloop_1 :
.Ls10 : movb ( % r s i ) ,% b l
.Ld10 : movb % b l ,( % r d i )
incq % r d i
incq % r s i
decl % e c x
jnz . L l o o p _ 1
2006-09-26 12:52:32 +04:00
CFI_ R E M E M B E R _ S T A T E
2006-02-03 23:51:02 +03:00
.Lende :
2006-09-26 12:52:39 +04:00
popq % r c x
CFI_ A D J U S T _ C F A _ O F F S E T - 8
CFI_ R E S T O R E r c x
2006-02-03 23:51:02 +03:00
popq % r b x
2006-09-26 12:52:32 +04:00
CFI_ A D J U S T _ C F A _ O F F S E T - 8
CFI_ R E S T O R E r b x
2006-02-03 23:51:02 +03:00
ret
2006-09-26 12:52:32 +04:00
CFI_ R E S T O R E _ S T A T E
2006-02-03 23:51:02 +03:00
# ifdef F I X _ A L I G N M E N T
/* align destination */
.p2align 4
.Lbad_alignment :
movl $ 8 ,% r9 d
subl % e c x ,% r9 d
movl % r9 d ,% e c x
cmpq % r9 ,% r d x
jz . L h a n d l e _ 7
js . L h a n d l e _ 7
.Lalign_1 :
.Ls11 : movb ( % r s i ) ,% b l
.Ld11 : movb % b l ,( % r d i )
incq % r s i
incq % r d i
decl % e c x
jnz . L a l i g n _ 1
subq % r9 ,% r d x
jmp . L a f t e r _ b a d _ a l i g n m e n t
# endif
/* table sorted by exception address */
.section _ _ ex_ t a b l e ," a "
.align 8
x86-64: Fix "bytes left to copy" return value for copy_from_user()
Most users by far do not care about the exact return value (they only
really care about whether the copy succeeded in its entirety or not),
but a few special core routines actually care deeply about exactly how
many bytes were copied from user space.
And the unrolled versions of the x86-64 user copy routines would
sometimes report that it had copied more bytes than it actually had.
Very few uses actually have partial copies to begin with, but to make
this bug even harder to trigger, most x86 CPU's use the "rep string"
instructions for normal user copies, and that version didn't have this
issue.
To make it even harder to hit, the one user of this that really cared
about the return value (and used the uncached version of the copy that
doesn't use the "rep string" instructions) was the generic write
routine, which pre-populated its source, once more hiding the problem by
avoiding the exception case that triggers the bug.
In other words, very special thanks to Bron Gondwana who not only
triggered this, but created a test-program to show it, and bisected the
behavior down to commit 08291429cfa6258c4cd95d8833beb40f828b194e ("mm:
fix pagecache write deadlocks") which changed the access pattern just
enough that you can now trigger it with 'writev()' with multiple
iovec's.
That commit itself was not the cause of the bug, it just allowed all the
stars to align just right that you could trigger the problem.
[ Side note: this is just the minimal fix to make the copy routines
(with __copy_from_user_inatomic_nocache as the particular version that
was involved in showing this) have the right return values.
We really should improve on the exceptional case further - to make the
copy do a byte-accurate copy up to the exact page limit that causes it
to fail. As it is, the callers have to do extra work to handle the
limit case gracefully. ]
Reported-by: Bron Gondwana <brong@fastmail.fm>
Cc: Nick Piggin <npiggin@suse.de>
Cc: Andrew Morton <akpm@linux-foundation.org>
Cc: Andi Kleen <andi@firstfloor.org>
Cc: Al Viro <viro@ZenIV.linux.org.uk>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
(which didn't have this problem), and since
most users that do the carethis was very hard to trigger, but
2008-06-18 04:47:50 +04:00
.quad .Ls1 , .Ls1e /* Ls1-Ls4 have copied zero bytes */
.quad .Ls2 , .Ls1e
.quad .Ls3 , .Ls1e
.quad .Ls4 , .Ls1e
.quad .Ld1 , .Ls1e /* Ld1-Ld4 have copied 0-24 bytes */
2006-02-03 23:51:02 +03:00
.quad .Ld2 , .Ls2e
.quad .Ld3 , .Ls3e
.quad .Ld4 , .Ls4e
x86-64: Fix "bytes left to copy" return value for copy_from_user()
Most users by far do not care about the exact return value (they only
really care about whether the copy succeeded in its entirety or not),
but a few special core routines actually care deeply about exactly how
many bytes were copied from user space.
And the unrolled versions of the x86-64 user copy routines would
sometimes report that it had copied more bytes than it actually had.
Very few uses actually have partial copies to begin with, but to make
this bug even harder to trigger, most x86 CPU's use the "rep string"
instructions for normal user copies, and that version didn't have this
issue.
To make it even harder to hit, the one user of this that really cared
about the return value (and used the uncached version of the copy that
doesn't use the "rep string" instructions) was the generic write
routine, which pre-populated its source, once more hiding the problem by
avoiding the exception case that triggers the bug.
In other words, very special thanks to Bron Gondwana who not only
triggered this, but created a test-program to show it, and bisected the
behavior down to commit 08291429cfa6258c4cd95d8833beb40f828b194e ("mm:
fix pagecache write deadlocks") which changed the access pattern just
enough that you can now trigger it with 'writev()' with multiple
iovec's.
That commit itself was not the cause of the bug, it just allowed all the
stars to align just right that you could trigger the problem.
[ Side note: this is just the minimal fix to make the copy routines
(with __copy_from_user_inatomic_nocache as the particular version that
was involved in showing this) have the right return values.
We really should improve on the exceptional case further - to make the
copy do a byte-accurate copy up to the exact page limit that causes it
to fail. As it is, the callers have to do extra work to handle the
limit case gracefully. ]
Reported-by: Bron Gondwana <brong@fastmail.fm>
Cc: Nick Piggin <npiggin@suse.de>
Cc: Andrew Morton <akpm@linux-foundation.org>
Cc: Andi Kleen <andi@firstfloor.org>
Cc: Al Viro <viro@ZenIV.linux.org.uk>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
(which didn't have this problem), and since
most users that do the carethis was very hard to trigger, but
2008-06-18 04:47:50 +04:00
.quad .Ls5 , .Ls5e /* Ls5-Ls8 have copied 32 bytes */
.quad .Ls6 , .Ls5e
.quad .Ls7 , .Ls5e
.quad .Ls8 , .Ls5e
.quad .Ld5 , .Ls5e /* Ld5-Ld8 have copied 32-56 bytes */
2006-02-03 23:51:02 +03:00
.quad .Ld6 , .Ls6e
.quad .Ld7 , .Ls7e
.quad .Ld8 , .Ls8e
.quad .Ls9 , .Le_quad
.quad .Ld9 , .Le_quad
.quad .Ls10 , .Le_byte
.quad .Ld10 , .Le_byte
# ifdef F I X _ A L I G N M E N T
.quad .Ls11 , .Lzero_rest
.quad .Ld11 , .Lzero_rest
# endif
.quad .Le5 , .Le_zero
.previous
/* eax: zero, ebx: 64 */
x86-64: Fix "bytes left to copy" return value for copy_from_user()
Most users by far do not care about the exact return value (they only
really care about whether the copy succeeded in its entirety or not),
but a few special core routines actually care deeply about exactly how
many bytes were copied from user space.
And the unrolled versions of the x86-64 user copy routines would
sometimes report that it had copied more bytes than it actually had.
Very few uses actually have partial copies to begin with, but to make
this bug even harder to trigger, most x86 CPU's use the "rep string"
instructions for normal user copies, and that version didn't have this
issue.
To make it even harder to hit, the one user of this that really cared
about the return value (and used the uncached version of the copy that
doesn't use the "rep string" instructions) was the generic write
routine, which pre-populated its source, once more hiding the problem by
avoiding the exception case that triggers the bug.
In other words, very special thanks to Bron Gondwana who not only
triggered this, but created a test-program to show it, and bisected the
behavior down to commit 08291429cfa6258c4cd95d8833beb40f828b194e ("mm:
fix pagecache write deadlocks") which changed the access pattern just
enough that you can now trigger it with 'writev()' with multiple
iovec's.
That commit itself was not the cause of the bug, it just allowed all the
stars to align just right that you could trigger the problem.
[ Side note: this is just the minimal fix to make the copy routines
(with __copy_from_user_inatomic_nocache as the particular version that
was involved in showing this) have the right return values.
We really should improve on the exceptional case further - to make the
copy do a byte-accurate copy up to the exact page limit that causes it
to fail. As it is, the callers have to do extra work to handle the
limit case gracefully. ]
Reported-by: Bron Gondwana <brong@fastmail.fm>
Cc: Nick Piggin <npiggin@suse.de>
Cc: Andrew Morton <akpm@linux-foundation.org>
Cc: Andi Kleen <andi@firstfloor.org>
Cc: Al Viro <viro@ZenIV.linux.org.uk>
Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org>
(which didn't have this problem), and since
most users that do the carethis was very hard to trigger, but
2008-06-18 04:47:50 +04:00
.Ls1e : addl $ 8 ,% e a x / * e a x i s b y t e s l e f t u n c o p i e d w i t h i n t h e l o o p ( L s1 e : 6 4 . . L s8 e : 8 ) * /
2006-02-03 23:51:02 +03:00
.Ls2e : addl $ 8 ,% e a x
.Ls3e : addl $ 8 ,% e a x
.Ls4e : addl $ 8 ,% e a x
.Ls5e : addl $ 8 ,% e a x
.Ls6e : addl $ 8 ,% e a x
.Ls7e : addl $ 8 ,% e a x
.Ls8e : addl $ 8 ,% e a x
addq % r b x ,% r d i / * + 6 4 * /
subq % r a x ,% r d i / * c o r r e c t d e s t i n a t i o n w i t h c o m p u t e d o f f s e t * /
shlq $ 6 ,% r d x / * l o o p c o u n t e r * 6 4 ( s t r i d e l e n g t h ) * /
addq % r a x ,% r d x / * a d d o f f s e t t o l o o p c n t * /
andl $ 6 3 ,% e c x / * r e m a i n i n g b y t e s * /
addq % r c x ,% r d x / * a d d t h e m * /
jmp . L z e r o _ r e s t
/* exception on quad word loop in tail handling */
/* ecx: loopcnt/8, %edx: length, rdi: correct */
.Le_quad :
shll $ 3 ,% e c x
andl $ 7 ,% e d x
addl % e c x ,% e d x
/* edx: bytes to zero, rdi: dest, eax:zero */
.Lzero_rest :
2006-09-26 12:52:39 +04:00
cmpl $ 0 ,( % r s p )
jz . L e _ z e r o
2006-02-03 23:51:02 +03:00
movq % r d x ,% r c x
.Le_byte :
xorl % e a x ,% e a x
.Le5 : rep
stosb
/* when there is another exception while zeroing the rest just return */
.Le_zero :
movq % r d x ,% r a x
jmp . L e n d e
2006-09-26 12:52:32 +04:00
CFI_ E N D P R O C
ENDPROC( c o p y _ u s e r _ g e n e r i c )
2006-02-03 23:51:02 +03:00
/ * Some C P U s r u n f a s t e r u s i n g t h e s t r i n g c o p y i n s t r u c t i o n s .
This i s a l s o a l o t s i m p l e r . U s e t h e m w h e n p o s s i b l e .
Patch i n j m p s t o t h i s c o d e i n s t e a d o f c o p y i n g i t f u l l y
to a v o i d u n w a n t e d a l i a s i n g i n t h e e x c e p t i o n t a b l e s . * /
/ * rdi d e s t i n a t i o n
* rsi s o u r c e
* rdx c o u n t
2006-09-26 12:52:39 +04:00
* ecx z e r o f l a g
2006-02-03 23:51:02 +03:00
*
* Output :
* eax u n c o p i e d b y t e s o r 0 i f s u c c e s s f u l l .
*
* Only 4 G B o f c o p y i s s u p p o r t e d . T h i s s h o u l d n ' t b e a p r o b l e m
* because t h e k e r n e l n o r m a l l y o n l y w r i t e s f r o m / t o p a g e s i z e d c h u n k s
* even i f u s e r s p a c e p a s s e d a l o n g e r b u f f e r .
* And m o r e w o u l d b e d a n g e r o u s b e c a u s e b o t h I n t e l a n d A M D h a v e
* errata w i t h r e p m o v s q > 4 G B . I f s o m e o n e f e e l s t h e n e e d t o f i x
* this p l e a s e c o n s i d e r t h i s .
2006-09-26 12:52:39 +04:00
* /
ENTRY( c o p y _ u s e r _ g e n e r i c _ s t r i n g )
2006-09-26 12:52:32 +04:00
CFI_ S T A R T P R O C
2006-09-26 12:52:39 +04:00
movl % e c x ,% r8 d / * s a v e z e r o f l a g * /
2005-04-17 02:20:36 +04:00
movl % e d x ,% e c x
shrl $ 3 ,% e c x
andl $ 7 ,% e d x
2006-09-26 12:52:39 +04:00
jz 1 0 f
2005-04-17 02:20:36 +04:00
1 : rep
movsq
movl % e d x ,% e c x
2 : rep
movsb
2006-09-26 12:52:39 +04:00
9 : movl % e c x ,% e a x
2005-04-17 02:20:36 +04:00
ret
2006-09-26 12:52:39 +04:00
/* multiple of 8 byte */
10 : rep
movsq
xor % e a x ,% e a x
2005-04-17 02:20:36 +04:00
ret
2006-09-26 12:52:39 +04:00
/* exception handling */
3 : lea ( % r d x ,% r c x ,8 ) ,% r a x / * e x c e p t i o n o n q u a d l o o p * /
jmp 6 f
5 : movl % e c x ,% e a x / * e x c e p t i o n o n b y t e l o o p * /
/* eax: left over bytes */
6 : testl % r8 d ,% r8 d / * z e r o f l a g s e t ? * /
jz 7 f
movl % e a x ,% e c x / * i n i t i a l i z e x86 l o o p c o u n t e r * /
push % r a x
xorl % e a x ,% e a x
8 : rep
stosb / * z e r o t h e r e s t * /
11 : pop % r a x
7 : ret
2006-09-26 12:52:32 +04:00
CFI_ E N D P R O C
END( c o p y _ u s e r _ g e n e r i c _ c )
2006-01-12 00:44:45 +03:00
2005-04-17 02:20:36 +04:00
.section _ _ ex_ t a b l e ," a "
.quad 1 b,3 b
2006-09-26 12:52:39 +04:00
.quad 2 b,5 b
.quad 8 b,1 1 b
.quad 1 0 b,3 b
2005-04-17 02:20:36 +04:00
.previous