deepset-ai · tholor · Oct 13, 2021 · Sep 1, 2021 · Sep 1, 2021 · Sep 13, 2021
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
@@ -13,7 +13,7 @@ jobs:
       - uses: actions/checkout@v2
       - uses: actions/setup-python@v2
         with:
-          python-version: 3.7
+          python-version: 3.8
       - name: Test with mypy
         run: |
           pip install mypy types-Markdown types-requests types-PyYAML

diff --git a/docs/_src/api/api/document_store.md b/docs/_src/api/api/document_store.md
@@ -63,7 +63,7 @@ Get documents from the document store.
 #### get\_all\_labels\_aggregated
 
 ```python
- | get_all_labels_aggregated(index: Optional[str] = None, filters: Optional[Dict[str, List[str]]] = None, open_domain: bool = True, aggregate_by_meta: Optional[Union[str, list]] = None) -> List[MultiLabel]
+ | get_all_labels_aggregated(index: Optional[str] = None, filters: Optional[Dict[str, List[str]]] = None, open_domain: bool = True, drop_negative_labels: bool = False, drop_no_answers: bool = False, aggregate_by_meta: Optional[Union[str, list]] = None) -> List[MultiLabel]
 ```
 
 Return all labels in the DocumentStore, aggregated into MultiLabel objects.
@@ -88,6 +88,7 @@ object, provided that they have the same product_id (to be found in Label.meta["
                     When False, labels are aggregated in a closed domain fashion based on the question text
                     and also the id of the document that the label is tied to. In this setting, this function
                     might return multiple MultiLabel objects with the same question string.
+:param TODO drop params
 - `aggregate_by_meta`: The names of the Label meta fields by which to aggregate. For example: ["product_id"]
 
 <a name="base.BaseDocumentStore.add_eval_data"></a>
@@ -131,7 +132,7 @@ class ElasticsearchDocumentStore(BaseDocumentStore)
 #### \_\_init\_\_
 
 ```python
- | __init__(host: Union[str, List[str]] = "localhost", port: Union[int, List[int]] = 9200, username: str = "", password: str = "", api_key_id: Optional[str] = None, api_key: Optional[str] = None, aws4auth=None, index: str = "document", label_index: str = "label", search_fields: Union[str, list] = "text", text_field: str = "text", name_field: str = "name", embedding_field: str = "embedding", embedding_dim: int = 768, custom_mapping: Optional[dict] = None, excluded_meta_data: Optional[list] = None, faq_question_field: Optional[str] = None, analyzer: str = "standard", scheme: str = "http", ca_certs: Optional[str] = None, verify_certs: bool = True, create_index: bool = True, refresh_type: str = "wait_for", similarity="dot_product", timeout=30, return_embedding: bool = False, duplicate_documents: str = 'overwrite', index_type: str = "flat")
+ | __init__(host: Union[str, List[str]] = "localhost", port: Union[int, List[int]] = 9200, username: str = "", password: str = "", api_key_id: Optional[str] = None, api_key: Optional[str] = None, aws4auth=None, index: str = "document", label_index: str = "label", search_fields: Union[str, list] = "content", content_field: str = "content", name_field: str = "name", embedding_field: str = "embedding", embedding_dim: int = 768, custom_mapping: Optional[dict] = None, excluded_meta_data: Optional[list] = None, analyzer: str = "standard", scheme: str = "http", ca_certs: Optional[str] = None, verify_certs: bool = True, create_index: bool = True, refresh_type: str = "wait_for", similarity="dot_product", timeout=30, return_embedding: bool = False, duplicate_documents: str = 'overwrite', index_type: str = "flat")
 ```
 
 A DocumentStore using Elasticsearch to store and query the documents for our search.
@@ -152,7 +153,7 @@ A DocumentStore using Elasticsearch to store and query the documents for our sea
 - `index`: Name of index in elasticsearch to use for storing the documents that we want to search. If not existing yet, we will create one.
 - `label_index`: Name of index in elasticsearch to use for storing labels. If not existing yet, we will create one.
 - `search_fields`: Name of fields used by ElasticsearchRetriever to find matches in the docs to our incoming query (using elastic's multi_match query), e.g. ["title", "full_text"]
-- `text_field`: Name of field that might contain the answer and will therefore be passed to the Reader Model (e.g. "full_text").
+- `content_field`: Name of field that might contain the answer and will therefore be passed to the Reader Model (e.g. "full_text").
                    If no Reader is used (e.g. in FAQ-Style QA) the plain content of this field will just be returned.
 - `name_field`: Name of field that contains the title of the the doc
 - `embedding_field`: Name of field containing an embedding vector (Only needed when using a dense retriever (e.g. DensePassageRetriever, EmbeddingRetriever) on top)
@@ -239,12 +240,12 @@ they will automatically get UUIDs assigned. See the `Document` class for details
 **Arguments**:
 
 - `documents`: a list of Python dictionaries or a list of Haystack Document objects.
-                  For documents as dictionaries, the format is {"text": "<the-actual-text>"}.
-                  Optionally: Include meta data via {"text": "<the-actual-text>",
+                  For documents as dictionaries, the format is {"content": "<the-actual-text>"}.
+                  Optionally: Include meta data via {"content": "<the-actual-text>",
                   "meta":{"name": "<some-document-name>, "author": "somebody", ...}}
                   It can be used for filtering and is accessible in the responses of the Finder.
                   Advanced: If you are using your own Elasticsearch mapping, the key names in the dictionary
-                  should be changed to what you have set for self.text_field and self.name_field.
+                  should be changed to what you have set for self.content_field and self.name_field.
 - `index`: Elasticsearch index where the documents should be indexed. If not supplied, self.index will be used.
 - `batch_size`: Number of documents that are passed to Elasticsearch's bulk function at a time.
 - `duplicate_documents`: Handle duplicates document based on parameter options.
@@ -1561,7 +1562,7 @@ The current implementation is not supporting the storage of labels, so you canno
 #### \_\_init\_\_
 
 ```python
- | __init__(host: Union[str, List[str]] = "http://localhost", port: Union[int, List[int]] = 8080, timeout_config: tuple = (5, 15), username: str = None, password: str = None, index: str = "Document", embedding_dim: int = 768, text_field: str = "text", name_field: str = "name", faq_question_field="question", similarity: str = "dot_product", index_type: str = "hnsw", custom_schema: Optional[dict] = None, return_embedding: bool = False, embedding_field: str = "embedding", progress_bar: bool = True, duplicate_documents: str = 'overwrite', **kwargs, ,)
+ | __init__(host: Union[str, List[str]] = "http://localhost", port: Union[int, List[int]] = 8080, timeout_config: tuple = (5, 15), username: str = None, password: str = None, index: str = "Document", embedding_dim: int = 768, content_field: str = "content", name_field: str = "name", similarity: str = "dot_product", index_type: str = "hnsw", custom_schema: Optional[dict] = None, return_embedding: bool = False, embedding_field: str = "embedding", progress_bar: bool = True, duplicate_documents: str = 'overwrite', **kwargs, ,)
 ```
 
 **Arguments**:
@@ -1574,10 +1575,9 @@ The current implementation is not supporting the storage of labels, so you canno
 - `password`: password (standard authentication via http_auth)
 - `index`: Index name for document text, embedding and metadata (in Weaviate terminology, this is a "Class" in Weaviate schema).
 - `embedding_dim`: The embedding vector size. Default: 768.
-- `text_field`: Name of field that might contain the answer and will therefore be passed to the Reader Model (e.g. "full_text").
+- `content_field`: Name of field that might contain the answer and will therefore be passed to the Reader Model (e.g. "full_text").
                    If no Reader is used (e.g. in FAQ-Style QA) the plain content of this field will just be returned.
 - `name_field`: Name of field that contains the title of the the doc
-- `faq_question_field`: Name of field containing the question in case of FAQ-Style QA
 - `similarity`: The similarity function used to compare document vectors. 'dot_product' is the default.
 - `index_type`: Index type of any vector object defined in weaviate schema. The vector index type is pluggable.
                    Currently, HSNW is only supported.

diff --git a/docs/_src/api/api/evaluation.md b/docs/_src/api/api/evaluation.md
@@ -89,7 +89,7 @@ open vs closed domain eval (https://haystack.deepset.ai/tutorials/evaluation).
 #### run
 
 ```python
- | run(labels: List[Label], answers: List[dict], correct_retrieval: bool)
+ | run(labels: List[Label], answers: List[Answer], correct_retrieval: bool)
 ```
 
 Run this node on one sample and its labels

diff --git a/docs/_src/api/api/file_converter.md b/docs/_src/api/api/file_converter.md
@@ -279,7 +279,7 @@ Extract text from a .pdf file using the pdftotext library (https://www.xpdfreade
                  others if your doc contains special characters (e.g. German Umlauts, Cyrillic characters ...).
                  Note: With "UTF-8" we experienced cases, where a simple "fi" gets wrongly parsed as
                  "xef\xac\x81c" (see test cases). That's why we keep "Latin 1" as default here.
-                 (See list of available encodings by running `pdftotext -listencodings` in the terminal)
+                 (See list of available encodings by running `pdftotext -listenc` in the terminal)
 
 <a name="pdf.PDFToTextOCRConverter"></a>
 ## PDFToTextOCRConverter Objects

diff --git a/docs/_src/api/api/generator.md b/docs/_src/api/api/generator.md
@@ -75,7 +75,7 @@ i.e. the model can easily adjust to domain documents even after training has fin
 |            'meta': { 'doc_ids': [...],
 |                      'doc_scores': [80.42758 ...],
 |                      'doc_probabilities': [40.71379089355469, ...
-|                      'texts': ['Albert Einstein was a ...]
+|                      'content': ['Albert Einstein was a ...]
 |                      'titles': ['"Albert Einstein"', ...]
 |      }}]}
 ```
@@ -134,7 +134,7 @@ Generated answers plus additional infos in a dict like this:
 |            'meta': { 'doc_ids': [...],
 |                      'doc_scores': [80.42758 ...],
 |                      'doc_probabilities': [40.71379089355469, ...
-|                      'texts': ['Albert Einstein was a ...]
+|                      'content': ['Albert Einstein was a ...]
 |                      'titles': ['"Albert Einstein"', ...]
 |      }}]}
 ```

diff --git a/docs/_src/api/api/preprocessor.md b/docs/_src/api/api/preprocessor.md
@@ -137,7 +137,7 @@ If batch_size is set to None, this method will yield all documents and labels.
 #### convert\_files\_to\_dicts
 
 ```python
-convert_files_to_dicts(dir_path: str, clean_func: Optional[Callable] = None, split_paragraphs: bool = False) -> List[dict]
+convert_files_to_dicts(dir_path: str, clean_func: Optional[Callable] = None, split_paragraphs: bool = False, encoding: Optional[str] = None) -> List[dict]
 ```
 
 Convert all files(.txt, .pdf, .docx) in the sub-directories of the given path to Python dicts that can be written to a
@@ -148,6 +148,7 @@ Document Store.
 - `dir_path`: path for the documents to be written to the DocumentStore
 - `clean_func`: a custom cleaning function that gets applied to each doc (input: str, output:str)
 - `split_paragraphs`: split text in paragraphs.
+- `encoding`: character encoding to use when converting pdf documents.
 
 **Returns**:
 

diff --git a/docs/_src/api/api/reader.md b/docs/_src/api/api/reader.md
@@ -241,7 +241,7 @@ Returns a dict containing the following metrics:
 #### eval
 
 ```python
- | eval(document_store: BaseDocumentStore, device: Optional[str] = None, label_index: str = "label", doc_index: str = "eval_document", label_origin: str = "gold_label", calibrate_conf_scores: bool = False)
+ | eval(document_store: BaseDocumentStore, device: Optional[str] = None, label_index: str = "label", doc_index: str = "eval_document", label_origin: str = "gold-label", calibrate_conf_scores: bool = False)
 ```
 
 Performs evaluation on evaluation documents in the DocumentStore.

diff --git a/docs/_src/api/api/retriever.md b/docs/_src/api/api/retriever.md
@@ -39,7 +39,7 @@ Wrapper method used to time functions.
 #### eval
 
 ```python
- | eval(label_index: str = "label", doc_index: str = "eval_document", label_origin: str = "gold_label", top_k: int = 10, open_domain: bool = False, return_preds: bool = False) -> dict
+ | eval(label_index: str = "label", doc_index: str = "eval_document", label_origin: str = "gold-label", top_k: int = 10, open_domain: bool = False, return_preds: bool = False) -> dict
 ```
 
 Performs evaluation on the Retriever.
@@ -105,7 +105,7 @@ class ElasticsearchRetriever(BaseRetriever)
                         |                "should": [{"multi_match": {
                         |                    "query": ${query},                 // mandatory query placeholder
                         |                    "type": "most_fields",
-                        |                    "fields": ["text", "title"]}}],
+                        |                    "fields": ["content", "title"]}}],
                         |                "filter": [                                 // optional custom filters
                         |                    {"terms": {"year": ${years}}},
                         |                    {"terms": {"quarter": ${quarters}}},
@@ -430,7 +430,7 @@ class EmbeddingRetriever(BaseRetriever)
 **Arguments**:
 
 - `document_store`: An instance of DocumentStore from which to retrieve documents.
-- `embedding_model`: Local path or name of model in Hugging Face's model hub such as ``'deepset/sentence_bert'``
+- `embedding_model`: Local path or name of model in Hugging Face's model hub such as ``'sentence-transformers/all-MiniLM-L6-v2'``
 - `model_version`: The version of model to use from the HuggingFace model hub. Can be tag name, branch name, or commit hash.
 - `use_gpu`: Whether to use gpu or not
 - `model_format`: Name of framework that was used for saving the model. Options:

diff --git a/docs/_src/api/api/translator.md b/docs/_src/api/api/translator.md
@@ -15,7 +15,7 @@ Abstract class for a Translator component that translates either a query or a do
 
 ```python
  | @abstractmethod
- | translate(query: Optional[str] = None, documents: Optional[Union[List[Document], List[str], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None) -> Union[str, List[Document], List[str], List[Dict[str, Any]]]
+ | translate(query: Optional[str] = None, documents: Optional[Union[List[Document], List[Answer], List[str], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None) -> Union[str, List[Document], List[Answer], List[str], List[Dict[str, Any]]]
 ```
 
 Translate the passed query or a list of documents from language A to B.
@@ -24,7 +24,7 @@ Translate the passed query or a list of documents from language A to B.
 #### run
 
 ```python
- | run(query: Optional[str] = None, documents: Optional[Union[List[Document], List[str], List[Dict[str, Any]]]] = None, answers: Optional[Union[Dict[str, Any], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None)
+ | run(query: Optional[str] = None, documents: Optional[Union[List[Document], List[Answer], List[str], List[Dict[str, Any]]]] = None, answers: Optional[Union[Dict[str, Any], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None)
 ```
 
 Method that gets executed when this class is used as a Node in a Haystack Pipeline
@@ -89,7 +89,7 @@ They also have a few multilingual models that support multiple languages at once
 #### translate
 
 ```python
- | translate(query: Optional[str] = None, documents: Optional[Union[List[Document], List[str], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None) -> Union[str, List[Document], List[str], List[Dict[str, Any]]]
+ | translate(query: Optional[str] = None, documents: Optional[Union[List[Document], List[Answer], List[str], List[Dict[str, Any]]]] = None, dict_key: Optional[str] = None) -> Union[str, List[Document], List[Answer], List[str], List[Dict[str, Any]]]
 ```
 
 Run the actual translation. You can supply a query or a list of documents. Whatever is supplied will be translated.
@@ -98,5 +98,5 @@ Run the actual translation. You can supply a query or a list of documents. Whate
 
 - `query`: The query string to translate
 - `documents`: The documents to translate
-- `dict_key`: 
+- `dict_key`: If you pass a dictionary in `documents`, you can specify here the field which shall be translated.
 
diff --git a/docs/_src/tutorials/tutorials/13.md b/docs/_src/tutorials/tutorials/13.md
@@ -72,9 +72,9 @@ text1 = "Python is an interpreted, high-level, general-purpose programming langu
 text2 = "Princess Arya Stark is the third child and second daughter of Lord Eddard Stark and his wife, Lady Catelyn Stark. She is the sister of the incumbent Westerosi monarchs, Sansa, Queen in the North, and Brandon, King of the Andals and the First Men. After narrowly escaping the persecution of House Stark by House Lannister, Arya is trained as a Faceless Man at the House of Black and White in Braavos, using her abilities to avenge her family. Upon her return to Westeros, she exacts retribution for the Red Wedding by exterminating the Frey male line."
 text3 = "Dry Cleaning are an English post-punk band who formed in South London in 2018.[3] The band is composed of vocalist Florence Shaw, guitarist Tom Dowse, bassist Lewis Maynard and drummer Nick Buxton. They are noted for their use of spoken word primarily in lieu of sung vocals, as well as their unconventional lyrics. Their musical stylings have been compared to Wire, Magazine and Joy Division.[4] The band released their debut single, 'Magic of Meghan' in 2019. Shaw wrote the song after going through a break-up and moving out of her former partner's apartment the same day that Meghan Markle and Prince Harry announced they were engaged.[5] This was followed by the release of two EPs that year: Sweet Princess in August and Boundary Road Snacks and Drinks in October. The band were included as part of the NME 100 of 2020,[6] as well as DIY magazine's Class of 2020.[7] The band signed to 4AD in late 2020 and shared a new single, 'Scratchcard Lanyard'.[8] In February 2021, the band shared details of their debut studio album, New Long Leg. They also shared the single 'Strong Feelings'.[9] The album, which was produced by John Parish, was released on 2 April 2021.[10]"
 
-docs = [{"text": text1},
-        {"text": text2},
-        {"text": text3}]
+docs = [{"content": text1},
+        {"content": text2},
+        {"content": text3}]
 
 # Initialize document store and write in the documents
 document_store = ElasticsearchDocumentStore()

diff --git a/docs/_src/tutorials/tutorials/5.md b/docs/_src/tutorials/tutorials/5.md
@@ -119,7 +119,7 @@ document_store.add_eval_data(
 )
 
 # Let's prepare the labels that we need for the retriever and the reader
-labels = document_store.get_all_labels_aggregated(index=label_index)
+labels = document_store.get_all_labels_aggregated(index=label_index, drop_negative_labels=True, drop_no_answers=False)
 ```
 
 ## Initialize components of QA-System
@@ -220,7 +220,7 @@ results = []
 # This is how to run the pipeline
 for l in labels:
     res = p.run(
-        query=l.question,
+        query=l.query,
         labels=l,
         params={"index": doc_index, "Retriever": {"top_k": 10}, "Reader": {"top_k": 5}},
     )

diff --git a/docs/_src/tutorials/tutorials/7.md b/docs/_src/tutorials/tutorials/7.md
@@ -82,7 +82,7 @@ documents: List[Document] = []
 for title, text in zip(titles, texts):
     documents.append(
         Document(
-            text=text,
+            content=text,
             meta={
                 "name": title or ""
             }