Improve document extraction and including a document extraction command (#17183)
* Add extract documents content command * Adding the extraction command and making the pure go pdf library as secondary option * Improving the memory usage and docextractor interface * Enable content extraction by default in all the instances * Tiny improvement on archive indexing * Adding App interface generation and the opentracing layer * Fixing linter errors * Addressing PR review comments * Addressing PR review comments
Этот коммит содержится в:
коммит произвёл
GitHub
родитель
75824257d5
Коммит
819e4c0c64
@@ -14,7 +14,6 @@ import (
|
||||
|
||||
"github.com/mattermost/mattermost-server/v5/model"
|
||||
"github.com/mattermost/mattermost-server/v5/plugin"
|
||||
"github.com/mattermost/mattermost-server/v5/services/docextractor"
|
||||
"github.com/mattermost/mattermost-server/v5/shared/mlog"
|
||||
"github.com/mattermost/mattermost-server/v5/store"
|
||||
)
|
||||
@@ -303,15 +302,9 @@ func (a *App) UploadData(us *model.UploadSession, rd io.Reader) (*model.FileInfo
|
||||
if *a.Config().FileSettings.ExtractContent && a.Config().FeatureFlags.FilesSearch {
|
||||
infoCopy := *info
|
||||
a.Srv().Go(func() {
|
||||
text, err := docextractor.Extract(infoCopy.Name, file, docextractor.ExtractSettings{
|
||||
ArchiveRecursion: *a.Config().FileSettings.ArchiveRecursion,
|
||||
})
|
||||
err := a.ExtractContentFromFileInfo(&infoCopy)
|
||||
if err != nil {
|
||||
mlog.Error("Failed to extract file content", mlog.Err(err))
|
||||
return
|
||||
}
|
||||
if storeErr := a.Srv().Store.FileInfo().SetContent(infoCopy.Id, text); storeErr != nil {
|
||||
mlog.Error("Failed to save the extracted file content", mlog.Err(storeErr))
|
||||
mlog.Error("Failed to extract file content", mlog.Err(err), mlog.String("fileInfoId", infoCopy.Id))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
Ссылка в новой задаче
Block a user