Fix performance problem on document extraction (#17470)
Этот коммит содержится в:
коммит произвёл
GitHub
родитель
dd9a1918e1
Коммит
bf542ec12f
@@ -20,10 +20,15 @@ func (pe *plainExtractor) Extract(filename string, r io.ReadSeeker) (string, err
|
|||||||
// This detects any visible character plus any whitespace
|
// This detects any visible character plus any whitespace
|
||||||
validRanges := append(unicode.GraphicRanges, unicode.White_Space)
|
validRanges := append(unicode.GraphicRanges, unicode.White_Space)
|
||||||
|
|
||||||
text, _ := ioutil.ReadAll(r)
|
runes := make([]byte, 1028)
|
||||||
|
_, err := r.Read(runes)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
count := 0
|
count := 0
|
||||||
for {
|
for {
|
||||||
c, size := utf8.DecodeRune(text[count:])
|
c, size := utf8.DecodeRune(runes[count:])
|
||||||
if !unicode.In(c, validRanges...) {
|
if !unicode.In(c, validRanges...) {
|
||||||
return "", nil
|
return "", nil
|
||||||
}
|
}
|
||||||
@@ -36,5 +41,6 @@ func (pe *plainExtractor) Extract(filename string, r io.ReadSeeker) (string, err
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
text, _ := ioutil.ReadAll(r)
|
||||||
return string(text), nil
|
return string(text), nil
|
||||||
}
|
}
|
||||||
|
|||||||
Ссылка в новой задаче
Block a user