Description
Pretrained BGEM3Embeddings model, adapted from Hugging Face and curated to provide scalability and production-readiness using Spark NLP. bge_m3 is a multilingual model originally trained by BAAI.
Predicted Entities
How to use
from sparknlp.base import *
from sparknlp.annotator import *
from pyspark.ml import Pipeline
document_assembler = DocumentAssembler() \
.setInputCol("text") \
.setOutputCol("document")
embeddings = BGEM3Embeddings.pretrained("bge_m3", "xx") \
.setInputCols(["document"]) \
.setOutputCol("embeddings") \
.setReturnSparseEmbeddings(True)
nlp_pipeline = Pipeline(stages=[
document_assembler,
embeddings
])
data = spark.createDataFrame([
["What is BGE M3?"],
["BGE M3 ist ein multilinguales Embedding-Modell."]
]).toDF("text")
result = nlp_pipeline.fit(data).transform(data)
result.selectExpr(
"text",
"embeddings.embeddings as dense",
"embeddings.metadata as sparse"
).show(truncate=60)
import com.johnsnowlabs.nlp.base._
import com.johnsnowlabs.nlp.embeddings._
import org.apache.spark.ml.Pipeline
val documentAssembler = new DocumentAssembler()
.setInputCol("text")
.setOutputCol("document")
val embeddings = BGEM3Embeddings.pretrained("bge_m3", "xx")
.setInputCols(Array("document"))
.setOutputCol("embeddings")
.setReturnSparseEmbeddings(true)
val pipeline = new Pipeline().setStages(Array(
documentAssembler,
embeddings
))
val data = Seq(
"What is BGE M3?",
"BGE M3 ist ein multilinguales Embedding-Modell."
).toDF("text")
val result = pipeline.fit(data).transform(data)
result.selectExpr(
"text",
"embeddings.embeddings as dense",
"embeddings.metadata as sparse"
).show(truncate = 60)
Model Information
| Model Name: | bge_m3 |
| Compatibility: | Spark NLP 7.0.0+ |
| License: | Open Source |
| Edition: | Official |
| Input Labels: | [document] |
| Output Labels: | [embeddings] |
| Language: | xx |
| Size: | 1.3 GB |