Explorar o código

num parallel embed

Roy Han hai 9 meses
pai
achega
2647a0e443
Modificáronse 1 ficheiros con 2 adicións e 0 borrados
  1. 2 0
      server/sched.go

+ 2 - 0
server/sched.go

@@ -132,6 +132,8 @@ func (s *Scheduler) processPending(ctx context.Context) {
 			if len(pending.model.ProjectorPaths) > 0 && numParallel != 1 {
 				numParallel = 1
 				slog.Warn("multimodal models don't support parallel requests yet")
+			} else if strings.Contains(pending.model.Config.ModelFamily, "bert") {
+				numParallel = runtime.NumCPU()
 			}
 
 			for {