Modified post handler for filtering out tags and poses
892
Flask based REST API for spaCy, the great and fast NLP framework. Supports the English and German language models and returns the analysis structured by sentences and by token.
Please note that currently the dependency trees and word vectors are not being returned.
curl http://localhost:5000/api --header 'content-type: application/json' --data '{"text": "This is a text that I want to be analyzed."}' -X POST
You'll receive a JSON in return:
{
'sentences': [[TOKEN, TOKEN, ...], [TOKEN, TOKEN, ...], ...],
'performance': CALCULATION_TIME_IN_SEC,
'version': SPACY_VERSION,
'numOfSentences': NUM_OF_SENTENCES,
'numOfTokens': NUM_OF_TOKENS
}
TOKEN: {
'token': TOKEN,
'lemma': LEMMA,
'tag': TAG,
'ner': NER,
'offsets': {
'begin': BEGIN,
'end': END
},
'oov': OUT_OF_VOCAB,
'stop': IS_STOPWORD,
'url': IS_URL,
'email': IS_MAIL,
'num': IS_NUM,
'pos': POS
}
| Field | Explanation |
|---|---|
| text | One text to be analyzed |
| texts | List of texts to be analyzed |
| fields | Optional. A list of token data fields that should be analyzed. Example: ['pos', 'token'] |
Either 'text' or 'texts' is required.
docker pull jgontrum/spacyapi:en
or
docker pull jgontrum/spacyapi:de
make english
or
make german
docker run --name spacyapi -d -p 127.0.0.1:5000:5000 jgontrum/spacyapi:en
make run-en
or
make run-de
curl http://localhost:5000/api --header 'content-type: application/json' --data '{"text": "Das hier ist Peter. Peter ist eine Person."}' -X POST
{
"performance": 0.0042879581451416016,
"version": "1.2.0",
"numOfSentences": 2,
"numOfTokens": 10,
"sentences": [
[
{
"offsets": {
"begin": 0,
"end": 3
},
"oov": false,
"stop": false,
"pos": "PRON",
"tag": "PDS",
"url": false,
"lemma": "das",
"token": "Das",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 4,
"end": 8
},
"oov": false,
"stop": false,
"pos": "ADV",
"tag": "ADV",
"url": false,
"lemma": "hier",
"token": "hier",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 9,
"end": 12
},
"oov": false,
"stop": false,
"pos": "AUX",
"tag": "VAFIN",
"url": false,
"lemma": "ist",
"token": "ist",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 13,
"end": 18
},
"oov": false,
"stop": false,
"pos": "PROPN",
"tag": "NE",
"url": false,
"lemma": "peter",
"token": "Peter",
"num": false,
"ner": "PERSON",
"email": false
},
{
"offsets": {
"begin": 18,
"end": 19
},
"oov": false,
"stop": false,
"pos": "PUNCT",
"tag": "$.",
"url": false,
"lemma": ".",
"token": ".",
"num": false,
"ner": "",
"email": false
}
],
[
{
"offsets": {
"begin": 20,
"end": 25
},
"oov": false,
"stop": false,
"pos": "PROPN",
"tag": "NE",
"url": false,
"lemma": "peter",
"token": "Peter",
"num": false,
"ner": "PERSON",
"email": false
},
{
"offsets": {
"begin": 26,
"end": 29
},
"oov": false,
"stop": false,
"pos": "AUX",
"tag": "VAFIN",
"url": false,
"lemma": "ist",
"token": "ist",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 30,
"end": 34
},
"oov": false,
"stop": false,
"pos": "DET",
"tag": "ART",
"url": false,
"lemma": "eine",
"token": "eine",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 35,
"end": 41
},
"oov": false,
"stop": false,
"pos": "NOUN",
"tag": "NN",
"url": false,
"lemma": "Person",
"token": "Person",
"num": false,
"ner": "",
"email": false
},
{
"offsets": {
"begin": 41,
"end": 42
},
"oov": false,
"stop": false,
"pos": "PUNCT",
"tag": "$.",
"url": false,
"lemma": ".",
"token": ".",
"num": false,
"ner": "",
"email": false
}
]
]
}
curl -X POST \
http://131.159.30.9:5000/api \
-H 'cache-control: no-cache' \
-H 'content-type: application/json' \
-H 'postman-token: c4375240-f4c6-cad5-f1cb-e0f5d129e83d' \
-d ' {"text": "In branch-2.8 and later, the patches for various child and related bugs listed in HADOOP-10105, most recently including HADOOP-11613, HADOOP-12710, HADOOP-12711, HADOOP-12552, and HDFS-10623, eliminate all use of \"commons-httpclient\" from Hadoop and its sub-projects (except for hadoop-tools/hadoop-openstack; see HADOOP-11614). However, after incorporating these patches, \"commons-httpclient\" is still listed as a dependency in these POM files:*hadoop-project/pom.xm* hadoop-yarn-project/hadoop-yarn/hadoop-yarn-registry/pom.xml We wish to remove these, but since commons-httpclient is still used in many files in hadoop-tools/hadoop-openstack, we'\''ll need to _add_ the dependency to * hadoop-tools/hadoop-openstack/pom.xml\r\n(We'\''ll add a note to HADOOP-11614 to undo this when commons-httpclient is removed from hadoop-openstack. In 2.8, this was mostly done by HADOOP-12552, but the version info formerly inherited from hadoop-project/pom.xml also needs to be added, so that is in the branch-2.8 version of the patch. Other projects with undeclared transitive dependencies on commons-httpclient, previously provided via hadoop-common or hadoop-client, may find this to be an incompatible change. Of course that also means such project is exposed to the commons-httpclient CVE, and needs to be fixed for that reason as well.", "fields":["offsets", "token", "pos", "tag", "deps"], "tags":["MD", "JJS", "JJR", "RBS", "RBR", "NN"]}
'
{
"numOfSentences": 5,
"error": false,
"lang": "en",
"version": "1.2.0",
"performance": [
0.027321338653564453
],
"numOfTokens": 44,
"data": [
{
"token": "branch-2.8",
"pos": "NOUN",
"deps": [
{
"dep_type": "pobj",
"token": "branch-2.8",
"tag": "NN",
"offsets": {
"end": 13,
"begin": 3
}
},
{
"dep_type": "cc",
"token": "and",
"tag": "CC",
"offsets": {
"end": 17,
"begin": 14
}
}
],
"tag": "NN"
},
{
"token": "child",
"pos": "NOUN",
"deps": [
{
"dep_type": "amod",
"token": "various",
"tag": "JJ",
"offsets": {
"end": 48,
"begin": 41
}
},
{
"dep_type": "pobj",
"token": "child",
"tag": "NN",
"offsets": {
"end": 54,
"begin": 49
}
},
{
"dep_type": "cc",
"token": "and",
"tag": "CC",
"offsets": {
"end": 58,
"begin": 55
}
},
{
"dep_type": "amod",
"token": "related",
"tag": "JJ",
"offsets": {
"end": 66,
"begin": 59
}
},
{
"dep_type": "conj",
"token": "bugs",
"tag": "NNS",
"offsets": {
"end": 71,
"begin": 67
}
},
{
"dep_type": "acl",
"token": "listed",
"tag": "VBN",
"offsets": {
"end": 78,
"begin": 72
}
},
{
"dep_type": "prep",
"token": "in",
"tag": "IN",
"offsets": {
"end": 81,
"begin": 79
}
},
{
"dep_type": "pobj",
"token": "HADOOP-10105",
"tag": "NNP",
"offsets": {
"end": 94,
"begin": 82
}
}
],
"tag": "NN"
},
{
"token": "most",
"pos": "ADV",
"deps": [
{
"dep_type": "advmod",
"token": "most",
"tag": "RBS",
"offsets": {
"end": 100,
"begin": 96
}
}
],
"tag": "RBS"
},
{
"token": "use",
"pos": "NOUN",
"deps": [
{
"dep_type": "det",
"token": "all",
"tag": "DT",
"offsets": {
"end": 205,
"begin": 202
}
},
{
"dep_type": "dobj",
"token": "use",
"tag": "NN",
"offsets": {
"end": 209,
"begin": 206
}
},
{
"dep_type": "prep",
"token": "of",
"tag": "IN",
"offsets": {
"end": 212,
"begin": 210
}
},
{
"dep_type": "punct",
"token": "\"",
"tag": "``",
"offsets": {
"end": 214,
"begin": 213
}
},
{
"dep_type": "amod",
"token": "commons",
"tag": "NNS",
"offsets": {
"end": 221,
"begin": 214
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 222,
"begin": 221
}
},
{
"dep_type": "pobj",
"token": "httpclient",
"tag": "NN",
"offsets": {
"end": 232,
"begin": 222
}
},
{
"dep_type": "punct",
"token": "\"",
"tag": "''",
"offsets": {
"end": 233,
"begin": 232
}
},
{
"dep_type": "prep",
"token": "from",
"tag": "IN",
"offsets": {
"end": 238,
"begin": 234
}
},
{
"dep_type": "pobj",
"token": "Hadoop",
"tag": "NNP",
"offsets": {
"end": 245,
"begin": 239
}
},
{
"dep_type": "cc",
"token": "and",
"tag": "CC",
"offsets": {
"end": 249,
"begin": 246
}
},
{
"dep_type": "poss",
"token": "its",
"tag": "PRP$",
"offsets": {
"end": 253,
"begin": 250
}
},
{
"dep_type": "compound",
"token": "sub",
"tag": "NN",
"offsets": {
"end": 257,
"begin": 254
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 258,
"begin": 257
}
},
{
"dep_type": "conj",
"token": "projects",
"tag": "NNS",
"offsets": {
"end": 266,
"begin": 258
}
},
{
"dep_type": "punct",
"token": "(",
"tag": "-LRB-",
"offsets": {
"end": 268,
"begin": 267
}
},
{
"dep_type": "prep",
"token": "except",
"tag": "IN",
"offsets": {
"end": 274,
"begin": 268
}
},
{
"dep_type": "prep",
"token": "for",
"tag": "IN",
"offsets": {
"end": 278,
"begin": 275
}
},
{
"dep_type": "nmod",
"token": "hadoop",
"tag": "NN",
"offsets": {
"end": 285,
"begin": 279
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 286,
"begin": 285
}
},
{
"dep_type": "compound",
"token": "tools/hadoop",
"tag": "NN",
"offsets": {
"end": 298,
"begin": 286
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 299,
"begin": 298
}
},
{
"dep_type": "pobj",
"token": "openstack",
"tag": "NN",
"offsets": {
"end": 308,
"begin": 299
}
}
],
"tag": "NN"
},
{
"token": "httpclient",
"pos": "NOUN",
"deps": [
{
"dep_type": "punct",
"token": "\"",
"tag": "``",
"offsets": {
"end": 214,
"begin": 213
}
},
{
"dep_type": "amod",
"token": "commons",
"tag": "NNS",
"offsets": {
"end": 221,
"begin": 214
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 222,
"begin": 221
}
},
{
"dep_type": "pobj",
"token": "httpclient",
"tag": "NN",
"offsets": {
"end": 232,
"begin": 222
}
},
{
"dep_type": "punct",
"token": "\"",
"tag": "''",
"offsets": {
"end": 233,
"begin": 232
}
},
{
"dep_type": "prep",
"token": "from",
"tag": "IN",
"offsets": {
"end": 238,
"begin": 234
}
},
{
"dep_type": "pobj",
"token": "Hadoop",
"tag": "NNP",
"offsets": {
"end": 245,
"begin": 239
}
},
{
"dep_type": "cc",
"token": "and",
"tag": "CC",
"offsets": {
"end": 249,
"begin": 246
}
},
{
"dep_type": "poss",
"token": "its",
"tag": "PRP$",
"offsets": {
"end": 253,
"begin": 250
}
},
{
"dep_type": "compound",
"token": "sub",
"tag": "NN",
"offsets": {
"end": 257,
"begin": 254
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 258,
"begin": 257
}
},
{
"dep_type": "conj",
"token": "projects",
"tag": "NNS",
"offsets": {
"end": 266,
"begin": 258
}
},
{
"dep_type": "punct",
"token": "(",
"tag": "-LRB-",
"offsets": {
"end": 268,
"begin": 267
}
},
{
"dep_type": "prep",
"token": "except",
"tag": "IN",
"offsets": {
"end": 274,
"begin": 268
}
},
{
"dep_type": "prep",
"token": "for",
"tag": "IN",
"offsets": {
"end": 278,
"begin": 275
}
},
{
"dep_type": "nmod",
"token": "hadoop",
"tag": "NN",
"offsets": {
"end": 285,
"begin": 279
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 286,
"begin": 285
}
},
{
"dep_type": "compound",
"token": "tools/hadoop",
"tag": "NN",
"offsets": {
"end": 298,
"begin": 286
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 299,
"begin": 298
}
},
{
"dep_type": "pobj",
"token": "openstack",
"tag": "NN",
"offsets": {
"end": 308,
"begin": 299
}
}
],
"tag": "NN"
},
{
"token": "sub",
"pos": "NOUN",
"deps": [
{
"dep_type": "compound",
"token": "sub",
"tag": "NN",
"offsets": {
"end": 257,
"begin": 254
}
}
],
"tag": "NN"
},
{
"token": "hadoop",
"pos": "NOUN",
"deps": [
{
"dep_type": "nmod",
"token": "hadoop",
"tag": "NN",
"offsets": {
"end": 285,
"begin": 279
}
}
],
"tag": "NN"
},
{
"token": "tools/hadoop",
"pos": "NOUN",
"deps": [
{
"dep_type": "compound",
"token": "tools/hadoop",
"tag": "NN",
"offsets": {
"end": 298,
"begin": 286
}
}
],
"tag": "NN"
},
{
"token": "openstack",
"pos": "NOUN",
"deps": [
{
"dep_type": "nmod",
"token": "hadoop",
"tag": "NN",
"offsets": {
"end": 285,
"begin": 279
}
},
{
"dep_type": "punct",
"token": "-",
"tag": "HYPH",
"offsets": {
"end": 286,
"begin": 285
}
},
{
"dep_type": "compound",
"token": "tools/hadoop",
"tag": "NN",
Content type
Image
Digest
Size
1.3 GB
Last updated
almost 9 years ago
docker pull klimksh/spacy-api-docker