diff --git a/notebooks/functional/2/Reading_ConTextItems.ipynb b/notebooks/functional/2/Reading_ConTextItems.ipynb deleted file mode 100644 index 996578e..0000000 --- a/notebooks/functional/2/Reading_ConTextItems.ipynb +++ /dev/null @@ -1,168 +0,0 @@ -{ - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### This notebook generates how to generate ConTextItems" - ] - }, - { - "cell_type": "code", - "execution_count": 1, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [ - "import os\n", - "import pyConTextNLP.functional.conTextItem as CI\n" - ] - }, - { - "cell_type": "code", - "execution_count": 3, - "metadata": { - "collapsed": false - }, - "outputs": [ - { - "data": { - "text/plain": [ - "'/Users/brian/anaconda/envs/NLP/lib/python2.7/site-packages/pyConTextNLP-0.6.0.9-py2.7.egg/pyConTextNLP/functional/conTextItem.pyc'" - ] - }, - "execution_count": 3, - "metadata": {}, - "output_type": "execute_result" - } - ], - "source": [ - "CI.__file__" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### ConTextItems can be read from the web" - ] - }, - { - "cell_type": "code", - "execution_count": 2, - "metadata": { - "collapsed": false - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<<\\b(examination|exam|study)\\b>>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n" - ] - } - ], - "source": [ - "kb = [\"https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/lexical_kb_04292013.tsv\", \n", - " \"https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/criticalfinder_generalized_modifiers.tsv\"]\n", - "items = []\n", - "for k in kb:\n", - " items.extend(CI.readConTextItems(k)[0])\n", - "for i in items[0:10]:\n", - " print(i)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### ConTextItems can also be read from local files" - ] - }, - { - "cell_type": "code", - "execution_count": 5, - "metadata": { - "collapsed": false - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<<\\b(examination|exam|study)\\b>>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n", - "literal<>; category<>; re<>; rule<>\n" - ] - } - ], - "source": [ - "PCDIR = os.path.join(os.path.expanduser(\"~\"),\n", - " \"Documents\",\"NLP\",\"pyConTextNLP\")\n", - "kb_local = [\"https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/lexical_kb_04292013.tsv\",\n", - " os.path.join(PCDIR,\"KB\",\"quality_artifacts.tsv\")]\n", - "items_local = []\n", - "for k in kb_local:\n", - " items_local.extend(CI.readConTextItems(k)[0])\n", - "for i in items_local[0:10]:\n", - " print(i)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "collapsed": true - }, - "outputs": [], - "source": [] - } - ], - "metadata": { - "kernelspec": { - "display_name": "Python 2", - "language": "python", - "name": "python2" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 2 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython2", - "version": "2.7.9" - } - }, - "nbformat": 4, - "nbformat_minor": 0 -} diff --git a/pyConTextNLP/deprecated/itemData.py b/pyConTextNLP/deprecated/itemData.py deleted file mode 100644 index dec7853..0000000 --- a/pyConTextNLP/deprecated/itemData.py +++ /dev/null @@ -1,426 +0,0 @@ -#Copyright 2010 Brian E. Chapman -# -#Licensed under the Apache License, Version 2.0 (the "License"); -#you may not use this file except in compliance with the License. -#You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -#Unless required by applicable law or agreed to in writing, software -#distributed under the License is distributed on an "AS IS" BASIS, -#WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -#See the License for the specific language governing permissions and -#limitations under the License. - -""" -A module defining the itemData class. itemData objects are the basis tools for text markup. - -The module instantiates several instances of this object: - 1) probableNegations - 2) definiteNegations - 3) pseudoNegations - 4) indications - 5) historicals - 6) conjugates - 7) probables - 8) definites -""" -class itemData(list): - def __init__(self,*args): - super(itemData,self).__init__(*args) - self.__numEnteries = 4 - - def __validate(self,data): - """validate that data consists of the correct number of string arguments""" - try: - td = type(data) - if( td != type([]) and td != type(()) and td != type(set([])) ): - print "data not a valid container type",td - return False - if( len(data ) != self.__numEnteries ): - print "data must have %d elements."%self.__numEnteries - return False - for d in data: - if( type(d) != type('') ): - print "all data elements must be strings" - return False - - return True - except Exception, error: - print "failed in itemData.validate", error - return False - - def append(self,data): - if(self.__validate(data)): - super(itemData,self).append(data) - - def insert(self,index,data): - if(self.__validate(data)): - super(itemData,self).insert(index,data) - def extend(self,iterable): - for i in iterable: - if( not self.__validate(i) ): - return - super(itemData,self).extend(iterable) - -probableNegations = itemData([ -["can rule out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["cannot be excluded","PROBABLE_NEGATED_EXISTENCE",r"""(cannot|can\snot)\sbe\sexcluded""","backward"], -["is not excluded","PROBABLE_NEGATED_EXISTENCE",r"""(is|was|are|were)\snot\sexcluded""",'backward'], -["adequate to rule the patient out against","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule him out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not know of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no findings of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule her out against","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no suggestion of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule him out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule him out against","PROBABLE_NEGATED_EXISTENCE",r"""sufficient\sto\srule\s(him|her)\sout\sagainst""","forward"], -["not reveal","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule the patient out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule her out for","PROBABLE_NEGATED_EXISTENCE",r"""adequate\sto\srule\s(her|him)\sout\sfor""","forward"], -["sufficient to rule her out for","PROBABLE_NEGATED_EXISTENCE",r"""sufficient\sto\srule\s(her|him)\sout\sfor""","forward"], -["adequate to rule out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule the patient out against","PROBABLE_NEGATED_EXISTENCE",r"""can\srule\s(her|him)\sout\sagainst""","forward"], -["can rule him out against","PROBABLE_NEGATED_EXISTENCE",r"""can\srule\s(her|him)\sout\sagainst""","forward"], -["rather than","PROBABLE_NEGATED_EXISTENCE","","forward"], -["nothing","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not exhibit","PROBABLE_NEGATED_EXISTENCE","","forward"], -["checked for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["evaluate for","PROBABLE_NEGATED_EXISTENCE","","forward"], ### SHOULD THIS BE AN INDICATION? -["can rule the patient out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no findings to indicate","PROBABLE_NEGATED_EXISTENCE","","forward"], -["free","PROBABLE_NEGATED_EXISTENCE","","backward"], -["sufficient to rule him out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule out against","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no sign of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no definite","PROBABLE_NEGATED_EXISTENCE", r"""no[\s]*definite""","forward"], #fixes peco #188, #255 -["without sign of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule the patient out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule her out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule him out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no significant","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule him out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not feel","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no obvious","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no complaints of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not associated with","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule her out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not see","PROBABLE_NEGATED_EXISTENCE","","forward"], -["fails to reveal","PROBABLE_NEGATED_EXISTENCE","","forward"], -["to exclude","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule the patient out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["unremarkable for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not appreciate","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not complain of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not demonstrate","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not to be","PROBABLE_NEGATED_EXISTENCE","","forward"], -["unlikely","PROBABLE_NEGATED_EXISTENCE","","backward"], -["absence of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule her out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule him out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["denying","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no signs of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no abnormal","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule the patient out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["without indication of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule her out","PROBABLE_NEGATED_EXISTENCE","","forward"], -["adequate to rule the patient out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["cannot see","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no cause of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["no evidence of","PROBABLE_NEGATED_EXISTENCE","","forward"], -["not known to have","PROBABLE_NEGATED_EXISTENCE","","forward"], -["can rule her out against","PROBABLE_NEGATED_EXISTENCE",r"""can\srule\s(her|him)\sout\sagainst""","forward"], -["sufficient to rule her out for","PROBABLE_NEGATED_EXISTENCE",r"""sufficient\sto\srule\s(her|him)\sout\sfor""","forward"], -["sufficient to rule the patient out against","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule out against","PROBABLE_NEGATED_EXISTENCE","","forward"], -["sufficient to rule out for","PROBABLE_NEGATED_EXISTENCE","","forward"], -["test for","PROBABLE_NEGATED_EXISTENCE","","forward"], -]) - -definiteNegations = itemData([ -["deny","DEFINITE_NEGATED_EXISTENCE","","forward"], -["denied","DEFINITE_NEGATED_EXISTENCE","","forward"], -["denies","DEFINITE_NEGATED_EXISTENCE","","forward"], -["declined","DEFINITE_NEGATED_EXISTENCE","","forward"], -["declines","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled him out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule her out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["are ruled out","DEFINITE_NEGATED_EXISTENCE","","backward"], -["rules the patient out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled her out against","DEFINITE_NEGATED_EXISTENCE","""ruled\s(him|her)\sout\sagainst""","forward"], -["ruled her out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["rules him out for","DEFINITE_NEGATED_EXISTENCE",r"""rules\s(him|her)\sout\sfor""","forward"], -["rules out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["no other", "DEFINITE_NEGATED_EXISTENCE", r"""no[\s]*other""","forward"], #fixes pedoc #265 -["ruled the patient out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["is ruled out","DEFINITE_NEGATED_EXISTENCE","","backward"], -["did rule the patient out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled him out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["cannot","DEFINITE_NEGATED_EXISTENCE","","forward"], -["negative for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["is negative","DEFINITE_NEGATED_EXISTENCE",r"(is|was) negative","backward"], -["-ve for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["not","DEFINITE_NEGATED_EXISTENCE",r"\bnot\b","forward"], -["never had","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule the patient out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["patient was not","DEFINITE_NEGATED_EXISTENCE","","forward"], -["has been ruled out","DEFINITE_NEGATED_EXISTENCE","","backward"], -["rules her out for","DEFINITE_NEGATED_EXISTENCE",r"""rules\s(him|her)\sout\sfor""","forward"], -["with no","DEFINITE_NEGATED_EXISTENCE","","forward"], -["not had","DEFINITE_NEGATED_EXISTENCE","","forward"], -["rules him out","DEFINITE_NEGATED_EXISTENCE",r"""rules\s(him|her)\sout""","forward"], -["rules the patient out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule the patient out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule him out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled him out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["no new","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled her out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule her out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule him out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["free of","DEFINITE_NEGATED_EXISTENCE","","forward"], -["not have","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled the patient out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["negative examination for","DEFINITE_NEGATED_EXISTENCE",r"negative (examination|study|exam|evaluation) for","forward"], -["rules out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["rules her out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["resolved","DEFINITE_NEGATED_EXISTENCE","","backward"], -["ruled him out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule him out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["no","DEFINITE_NEGATED_EXISTENCE",r"\bno\b","forward"], -["was ruled out","DEFINITE_NEGATED_EXISTENCE","","backward"], -["did rule her out against","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["ruled the patient out for","DEFINITE_NEGATED_EXISTENCE","","forward"], -["never developed","DEFINITE_NEGATED_EXISTENCE","","forward"], -["did rule her out","DEFINITE_NEGATED_EXISTENCE","","forward"], -["without","DEFINITE_NEGATED_EXISTENCE","","forward"], -["have been ruled out","DEFINITE_NEGATED_EXISTENCE","","backward"], -]) - -pseudoNegations = itemData([ -["not only","PSEUDONEG","",""], -["no definite change","PSEUDONEG","","forward"], -["not cause","PSEUDONEG","","forward"], #should have a re for "not (the) cause" -["without difficulty","PSEUDONEG","","forward"], -["not extend","PSEUDONEG","","forward"], -["not necessarily","PSEUDONEG","","forward"], -["not certain whether","PSEUDONEG","","forward"], -["no significant change","PSEUDONEG","","forward"], -["no suspicious change","PSEUDONEG","","forward"], -["no increase","PSEUDONEG","","forward"], -["no significant interval change","PSEUDONEG","","forward"], -["not drain","PSEUDONEG","","forward"], -["gram negative","PSEUDONEG","","forward"], -["no change","PSEUDONEG","","forward"], -#["the examination","PSEUDONEG","","backward"], -#["positive study for","PSEDUONEG",r"positive (study|exam|examination)( for)","forward"], -]) - -indications = itemData([ -["will be ruled out","INDICATION","","backward"], -["can be ruled out","INDICATION","","backward"], -["will be ruled out for","INDICATION","","forward"], -["should be ruled out","INDICATION","","backward"], -#["did not rule out","INDICATION","","backward"], -["rule out","INDICATION",r"(r/o|rule out|\br o\b|r\.o\.|\bro\b)","forward"], -["could be ruled out","INDICATION","","backward"], -#["not ruled out","INDICATION","","backward"], -["ought to be ruled out for","INDICATION","","forward"], -["be ruled out","INDICATION","","backward"], -["could be ruled out for","INDICATION","","forward"], -["must be ruled out","INDICATION","","backward"], -["must be ruled out for","INDICATION","","forward"], -["ought to be ruled out","INDICATION","","backward"], -["can be ruled out for","INDICATION","","forward"], -["rule him out for","INDICATION","","forward"], -["may be ruled out for","INDICATION","","forward"], -["may be ruled out","INDICATION","","backward"], -["is to be ruled out for","INDICATION","","forward"], -["is to be ruled out","INDICATION","","backward"], -["rule him out","INDICATION","","forward"], -["might be ruled out for","INDICATION","","forward"], -["should be ruled out for","INDICATION","","forward"], -["rule her out","INDICATION","","forward"], -["not been ruled out","INDICATION","","backward"], -["might be ruled out","INDICATION","","backward"], -["rule patient out for","INDICATION",r"rule (him|her|patient|the patient|subject) out for","forward"], -["evaluation of","INDICATION",r"evaluation\s(of|for)","forward"], -["evaluation","INDICATION","","bidirectional"], -["being ruled out","INDICATION","","backward"], -["what must be ruled out is","INDICATION","","forward"], -["examination for","INDICATION",r"(study|exam|examination) for","forward"], -["study for detection","INDICATION","","forward"], -["examination","INDICATION",r"\b(examination|exam|study)\b","backward"], -["protocol",'INDICATION','','backward'], -["rule the patient out","INDICATION","","forward"], -["rule out for","INDICATION","","forward"], -["be ruled out for","INDICATION","","forward"], -["rule the patient out for","INDICATION","","forward"]]) -"""THIS IS AGE INDETERMINATE -SUBACUTE -RESIDUAL -RESOLUTION OF PRIOR -RESOLUTION OF -resolving -no change in -chaninging -PREVIOUSLY NOTED -UNCHANGED -HAS DIMINISHED -INTERVAL CHANGE IN - -""" -historicals = itemData([ -["documented","HISTORICAL","","forward"], -["subacute","HISTORICAL","","forward"], -["chronic","HISTORICAL","","forward"], -["previous","HISTORICAL","","forward"], # fixes pedoc #98, 115 -["resolving","HISTORICAL","","forward"], -["resolved","HISTORICAL","","backward"], -["previous","HISTORICAL","","forward"], -["interval change","HISTORICAL","","bidirectional"], -["resolution of","HISTORICAL","","forward"], -["clinical history","HISTORICAL","","forward"], -["unchanged","HISTORICAL","","bidirectional"], -["changing","HISTORICAL","","forward"], -["change in","HISTORICAL","","forward"], -["prior","HISTORICAL","","bidirectional"], -["diminished","HISTORICAL","","bidirectional"], -["sequelae of","HISTORICAL","","forward"], -["prior study", "HISTORICAL","","bidirectional"], -]) - -conjugates = itemData([ -#["with","CONJ","","terminate"], # fixes pedoc 131 scope -["involving","CONJ","","terminate"], # proposed fix for pedoc #153 -["as a secondary cause for","CONJ","","terminate"], -["as the secondary etiology for","CONJ","","terminate"], -["as a secondary source of","CONJ","","terminate"], -["as an etiology of","CONJ","","terminate"], -["as the secondary reason of","CONJ","","terminate"], -["as the secondary origin of","CONJ","","terminate"], -["as an secondary reason for","CONJ","","terminate"], -["as an secondary reason of","CONJ","","terminate"], -["reason for","CONJ","","terminate"], -["still","CONJ","","terminate"], -["source of","CONJ","","terminate"], -["except","CONJ","","terminate"], -["etiology of","CONJ","","terminate"], -["as a cause of","CONJ","","terminate"], -["as a source of","CONJ","","terminate"], -["as the secondary etiology of","CONJ","","terminate"], -["as an reason for","CONJ","","terminate"], -["as a etiology for","CONJ","","terminate"], -["as a secondary origin for","CONJ","","terminate"], -["etiology for","CONJ","","terminate"], -["reasons for","CONJ","","terminate"], -["as a secondary cause of","CONJ","","terminate"], -["aside from","CONJ","","terminate"], -["as the origin of","CONJ","","terminate"], -["though","CONJ","","terminate"], -["which","CONJ","","terminate"], -["cause of","CONJ","","terminate"], -["as the secondary cause for","CONJ","","terminate"], -["as a source for","CONJ","","terminate"], -["as an origin for","CONJ","","terminate"], -["as a secondary origin of","CONJ","","terminate"], -["as the etiology for","CONJ","","terminate"], -["other possibilities of","CONJ","","terminate"], -["as an etiology for","CONJ","","terminate"], -["origins for","CONJ","","terminate"], -["as the secondary reason for","CONJ","","terminate"], -["as the secondary origin for","CONJ","","terminate"], -["as an reason of","CONJ","","terminate"], -["origin for","CONJ","","terminate"], -["as a cause for","CONJ","","terminate"], -["however","CONJ","","terminate"], -["secondary to","CONJ","","terminate"], -["although","CONJ","","terminate"], -["as an secondary source of","CONJ","","terminate"], -["as an source of","CONJ","","terminate"], -["as an cause for","CONJ","","terminate"], -["as the secondary cause of","CONJ","","terminate"], -["as a secondary reason of","CONJ","","terminate"], -["as the etiology of","CONJ","","terminate"], -["as an source for","CONJ","","terminate"], -["as an secondary etiology of","CONJ","","terminate"], -["reasons of","CONJ","","terminate"], -["as an cause of","CONJ","","terminate"], -["as an secondary cause for","CONJ","","terminate"], -["as a reason of","CONJ","","terminate"], -["but","CONJ","","terminate"], -["as the secondary source of","CONJ","","terminate"], -["as a etiology of","CONJ","","terminate"], -["reason of","CONJ","","terminate"], -["causes for","CONJ","","terminate"], -["yet","CONJ","","terminate"], -["as a secondary etiology for","CONJ","","terminate"], -["as the origin for","CONJ","","terminate"], -["as the reason for","CONJ","","terminate"], -["trigger event for","CONJ","","terminate"], -["as the reason of","CONJ","","terminate"], -["cause for","CONJ","","terminate"], -["as a reason for","CONJ","","terminate"], -["as an secondary cause of","CONJ","","terminate"], -["sources of","CONJ","","terminate"], -["as the cause for","CONJ","","terminate"], -["as the source of","CONJ","","terminate"], -["as the source for","CONJ","","terminate"], -["origin of","CONJ","","terminate"], -["causes of","CONJ","","terminate"], -["sources for","CONJ","","terminate"], -["as a secondary source for","CONJ","","terminate"], -["apart from","CONJ","","terminate"], -["source for","CONJ","","terminate"], -["as an secondary origin for","CONJ","","terminate"], -["origins of","CONJ","","terminate"], -["as an origin of","CONJ","","terminate"], -["as an secondary source for","CONJ","","terminate"], -["nevertheless","CONJ","","terminate"], -["as the secondary source for","CONJ","","terminate"], -["as a secondary reason for","CONJ","","terminate"], -["as an secondary etiology for","CONJ","","terminate"], -["as the cause of","CONJ","","terminate"], -["as a secondary etiology of","CONJ","","terminate"], -["as an secondary origin of","CONJ","","terminate"]]) - -probables = itemData([ -["seen best","PROBABLE_EXISTENCE","",""], # fixes pedoc #126 uncertainty -["consistent with","PROBABLE_EXISTENCE","","forward"], -["evidence","PROBABLE_EXISTENCE","","forward"], -["suggestive","PROBABLE_EXISTENCE","","forward"], -#["not excluded", "POST-UNCERTAINTY",r"""not\sexcluded""","backward"], #fixes pedoc #139 -["appear","PROBABLE_EXISTENCE","\bappear\b","bidirectional"], # fixes pedoc #270 uncertainty -#["definite","DEFINITE EXISTENCE","","forward"], -["may represent","PROBABLE_EXISTENCE",r"""(may|might)\srepresent""","forward"], -["appears to be","PROBABLE_EXISTENCE","","forward"], -["compatible with","PROBABLE_EXISTENCE","","forward"], -["convincing","PROBABLE_EXISTENCE","","forward"], -["suggest","PROBABLE_EXISTENCE",r"\bsuggest\b","forward"], -["represents","PROBABLE_EXISTENCE","","forward"], -["certain if","UNCERTAINTY","","forward"], -["suspicious","PROBABLE_EXISTENCE","","forward"], -["seen","PROBABLE_EXISTENCE",r"(seen|visualized|observed)","backward"], -["noted","PROBABLE_EXISTENCE","","backward"], -["worrisome","PROBABLE_EXISTENCE","","forward"], -["identified","PROBABLE_EXISTENCE","","backward"], -["suspicous","PROBABLE_EXISTENCE","","forward"], -["likely","PROBABLE_EXISTENCE","","bidirectional"], -["versus","PROBABLE_EXISTENCE","","bidirectional"], -["equivocal","PROBABLE_EXISTENCE","",'bidirectional'] -]) - -definites = itemData([ -["positive examination for","DEFINITE_EXISTENCE","","forward"], # fixes pedoc #126 uncertainty -["obvious","DEFINITE_EXISTENCE","","forward"], -["definite","DEFINITE_EXISTENCE","","forward"], -["positive study for","PSEDUONEG",r"positive (study|exam|examination)( for)","forward"], -]) \ No newline at end of file diff --git a/pyConTextNLP/deprecated/pycontext.py b/pyConTextNLP/deprecated/pycontext.py deleted file mode 100644 index 43eca45..0000000 --- a/pyConTextNLP/deprecated/pycontext.py +++ /dev/null @@ -1,541 +0,0 @@ -#Copyright 2010 Brian E. Chapman -# -#Licensed under the Apache License, Version 2.0 (the "License"); -#you may not use this file except in compliance with the License. -#You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -#Unless required by applicable law or agreed to in writing, software -#distributed under the License is distributed on an "AS IS" BASIS, -#WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -#See the License for the specific language governing permissions and -#limitations under the License. -""" -This module contains three class definitions that are used in the pyConText -algorithm. The pyConText algorithm relies on regular expressions to identify -sub-texts of interest - -1) termObject: a class that describes terms of interest within the text -2) tagObject: a class inherited from termObject that describes modifiers -3) pycontext: a class that implements the context algorithm - -""" -import re -import copy - - -class termObject(object): - """ - A class that describes terms of interest in the text. - termObject is characterized by the following attributes - 1) The textual term being sought - 2) The categorty this term fits in (e.g. finding) - 3) The location of the term within the text being parsed - 4) What other terms modify this term - 5) What terms this term modifies - """ - def __init__(self,term, category,regExp,*args): - """ - term: textual term to define object - category: category for this term - regExp: a regular expressiont that is used to catch term and its - variants - """ - self.__term = term - self.__category = category.lower() - self.__start = 0 # where the term occurs in the sentence - self.__spanStart = 99 - self.__spanEnd = 77 - self.__regExp = regExp - self.__foundPhrase = '' - self.__modifiedBy = {} # another object that modifies this object - self.__modifies = {} # another object modified by this object - - def getBriefDescription(self): - return """%s (%s) <<%s>>"""%(self.getTerm(),self.getPhrase(),self.getCategory()) - def getTerm(self): - """returns the term defining this object""" - return self.__term - def getCategory(self): - """returns the category (e.g. CONJUNCTION) for this object""" - return self.__category - def setStart(self,start): - """set the start location for this object in the associated text. - Is this redudant with span???""" - self.__start = start - def getStart(self): - """return the start location for this object""" - return self.__start - def setSpan(self,span): - """set the span within the associated text for this object""" - self.__spanStart = span[0] - self.__spanEnd = span[1] - def getSpan(self): - """return the span within the associated text for this object""" - return self.__spanStart,self.__spanEnd - def setPhrase(self,phrase): - """set the actual matched phrase used to generate this object""" - self.__foundPhrase = phrase - def getPhrase(self): - """return the actual matched phrase used to generate this object""" - return self.__foundPhrase - def getModifiedBy(self): - """returns the object modifying this object. - Why do I only have one object as a modifier???""" - return self.__modifiedBy - def setModifiedBy(self,obj): - """sets the object modifying this object""" - self.__modifiedBy = obj - def dropModifies(self,obj): - """remove obj from the list of objects modified by the current object""" - try: - self.__modifies[obj.getCategory()].remove(obj) - except: - pass - def addModifies(self,obj): - """add obj to the list of objects modified by the current object. There - is currently no safety check that obj is valid""" - tmp = self.__modifies.get(obj.getCategory(),[]) - tmp.append(obj) - self.__modifies[obj.getCategory()] = tmp - - def dist(self,obj): - """returns the minimum distance from the current object and obj. - Distance is measured as current start to object end or current end to object start""" - return min(abs(self.__spanEnd-obj.__spanStart),abs(self.__spanStart-obj.__spanEnd)) - - def updateModifiedBy(self,obj): - """Adds obj to the collection of objects modified by self""" - modifier = self.__modifiedBy.get(obj.getCategory()) - - if( not modifier ): - self.__modifiedBy[obj.getCategory()] = obj - obj.addModifies(self) - else: - if( self.dist(obj) < self.dist(modifier) ): - modifier.dropModifies(self) - self.__modifiedBy[obj.getCategory()] = obj - obj.addModifies(self) - - - def __lt__(self,other): return self.__start < other.__start - def __le__(self,other): return self.__start <= other.__start - def __eq__(self,other): - return (self.__start == other.__start and - self.__spanStart == other.__spanStart and - self.__spanEnd == other.__spanEnd) - def __ne__(self,other): return self.__start != other.__start - def __gt__(self,other): return self.__start > other.__start - def __ge__(self,other): return self.__start >= other.__start - def encompasses(self,other): - """tests whether other is completely encompassed with the current object""" - if( self.__spanStart <= other.__spanStart and - self.__spanEnd >= other.__spanEnd ): - return True - else: - return False - def __str__(self): - txt = """term<<%s>>; category<<%s>>; span<<%d, %d>>; matched phrase<<%s>>"""%\ - (self.__term,self.__category,self.__spanStart,self.__spanEnd,self.__foundPhrase) - if( self.__modifiedBy): txt += "; Modified by<<%s>>"%self.renderModifiedBy() - return txt - def renderModifiedBy(self): - txt = '' - keys = self.__modifiedBy.keys() - for k in keys: - obj = self.__modifiedBy[k] - txt += " %s[%s]@%d"%(obj.getPhrase(),obj.getCategory(),obj.getStart()) - return txt - def getNumModifiers(self): - """return the number of ojbects modifying the current object""" - return len(self.__modifiedBy) - def getModifier(self,i=0): - """This function strikes me as being very problematic since keys are not - ordered""" - try: - keys = self.__modifiedBy.keys() - return self.__modifiedBy[keys[i]] - except: - return None - def isModifiedByOr(self,terms): - """"returns true if self is modified by any term in the list terms""" - if( not self.__modifiedBy ): - return False - - for t in terms: - if( self.isModifiedBy(t) ): - return True - return False - - - def isModifiedByAnd(self,terms): - """return True if self is modified by all items in the list terms""" - if( not self.__modifiedBy ): - return False - if( type(terms) == type('') ): - return self.isModifiedBy(terms) - for t in terms: - if( not self.isModifiedBy(t) ): - return False - return True - - def isModifiedBy(self,term, returnModifier=False): - """tests whether self is modified by term. Optionally return the actual modifier""" - if( not self.__modifiedBy ): - return False - if( type(term) != type('') ): - raise TypeError("term must be a string") - keys = self.__modifiedBy.keys() - for k in keys: - modifier = self.__modifiedBy[k] - if( term.lower() in modifier.getCategory().lower() ): - if( returnModifier): - return modifier - else: - return True - return False - - def getModifiers(self): - """ - returns the dictionary of objects modifying the current object. - I should probably return a copy - """ - - return self.__modifiedBy.values() -class tagObject(termObject): - """ - A class that describes and manages tag of interest in the text - Adds the concept of a rule to termObject as well as the concept of a scope - """ - def __init__(self,term,category,regexp,*args): - """ - term, category and regexp are the same as for termObject - *args contains the rule and scope for the object - """ - super(tagObject,self).__init__(term,category,regexp,args) - - self.__rule = args[0].lower() - self.__scope = list(args[1][:]) - self.__SCOPEUPDATED = False - def setScope(self): - """ - applies the objects own rule and span to modify the object's scope - Currently only "forward" and "backward" rules are implemented - """ - - if( 'forward' in self.__rule.lower() ): - self.__scope[0] = self.getSpan()[1] - elif( 'backward' in self.__rule.lower() ): - self.__scope[1] = self.getSpan()[0] - - def __str__(self): - txt = super(tagObject,self).__str__() - txt += """; rule<<%s>>; scope<<%d,%d>>"""%(self.__rule,self.__scope[0],self.__scope[1]) - return txt - def getScope(self): - return self.__scope - def getRule(self): - return self.__rule - - def limitScope(self,obj): - """If self and obj are of the same category or if obj has a rule of - 'terminate', use the span of obj to - update the scope of self""" - if( not self.getRule() or self.getRule()== 'terminate' or - (self.getCategory() != obj.getCategory() and obj.getRule() != 'terminate')): - return - if( 'forward' in self.__rule.lower() ): - if( obj > self ): - self.__scope[1] = min(self.__scope[1],obj.getSpan()[0]) - elif( 'backward' in self.__rule.lower() ): - if( obj < self ): - self.__scope[0] = max(self.__scope[0],obj.getSpan()[1]) - - def applyRule(self,term): - """applies self's rule to term. If the start of term lines within - the span of self, then term may be modified by self""" - if( not self.getRule() or self.getRule() == 'terminate'): - return - if(self.__scope[0] <= term.getStart() <= self.__scope[1]): - term.updateModifiedBy(self) - - -class pycontext(object): - """ - base class for context. - build around markedTargets a list of termObjects representing desired terms - found in text and markedModifiers, tagObjects found in the text - """ - # regular expressions for cleaning text - r1 = re.compile(r"""\W""") - r2 = re.compile(r"""\s+""") - r3 = re.compile(r"""\d""") - # regular expression for identifying word boundaries (used for more - # complex rule specifications - rb = re.compile(r"""\b""") - def __init__(self,txt=''): - """txt is the string to parse""" - # __archive is for multisentence text processing. A markup is done - # for each sentence and then put in the archives when the next sentence - # is processed - self.__archive = {} - self.__currentSentence = 0 - self.__rawTxt = txt - self.__txt = None - self.__markedTargets = [] - self.__markedModifiers = [] - self.__scope = None - self.__SCOPEUPDATED = False - - # regular expressions for finding text - self.res = {} - - - def reset(self): - """deletes all archived values and sets all class attributes to empty or - zero values - """ - self.__archive = {} - self.__markedTargets = [] - self.__markedModifiers = [] - self.__scope = None - self.__SCOPEUPDATED - self.__currentSentence = 0 - def commit(self): - """ - takes the values stored in current attributes and copies them to the - object archive - """ - # I'm not sure if I want to be using copy here - self.__archive[self.__currentSentence] = (copy.copy(self.__rawTxt), - copy.copy(self.__txt), - copy.copy(self.__markedTargets), - copy.copy(self.__markedModifiers), - copy.copy(self.__scope), - copy.copy(self.__SCOPEUPDATED)) - self.__currentSentence += 1 - self.setTxt() - def setSentence(self,num): - """ - set the current context to sentence num in the archive - """ - self.__rawTxt = copy.copy(self.__archive[num][0]) - self.__txt = copy.copy(self.__archive[num][1]) - self.__markedTargets = copy.copy(self.__archive[num][2]) - self.__markedModifiers = copy.copy(self.__archive[num][3]) - self.__scope = copy.copy(self.__archive[num][4]) - self.__SCOPEUPDATED = copy.copy(self.__archive[num]) - - - def setTxt(self,txt=''): - """ - sets the current txt to txt and resets the current attributes to empty - values, but does not modify the object archive - """ - self.__rawTxt = txt - self.__txt = None - self.__markedTargets = [] - self.__markedModifiers = [] - self.__scope = None - self.__SCOPEUPDATED = False - - def getText(self): - return self.__txt - def getNumberSentences(self): - return len(self.__archive) - def getCurrentSentenceNumber(self): - return self.__currentSentence - def incrementSentence(self): - pass - def getNumMarkedTargets(self): - return len(self.__markedTargets) - def getNumMarkedModifiers(self): - return len(self.__markedModifiers) - def getMarkedTarget(self,i=0): - try: - return self.__markedTargets[i] - except: - return None - def getMarkedModifier(self,i=0): - try: - return self.__markedModifiers[i] - except: - return None - def getCleanTxt(self,stripNumbers=False): - """Need to rename. applies the regular expression scrubbers to rawTxt""" - self.__txt = self.r1.sub(" ",self.__rawTxt) - self.__txt = self.r2.sub(" ",self.__txt) - if( stripNumbers ): - self.__txt = self.r3.sub("",self.__txt) - - self.__scope= (0,len(self.__txt)) - def __str__(self): - txt = '' - txt += self.renderMarkedTargets() - #txt += """TAGS""".center(60)+"\n\n" - for term in self.__markedModifiers: - txt += term.__str__()+"\n" - txt += "-"*60 - return txt - def renderMarkedTargets(self): - if( not self.__markedTargets ): - return '' - txt = """TERMS""".center(60)+"\n\n" - for term in self.__markedTargets: - txt += term.__str__()+"\n" - return txt - def renderMarkedModifiers(self): - if( not self.__markedModifiers ): - return '' - txt = """TAGS""".center(60)+"\n\n" - for term in self.__markedModifiers: - txt += term.__str__()+"\n" - return txt - - def updateScopes(self): - """ - update the scopes of all the marked modifiers in the txt. The scope - of a modifier is limited by its own span and the and the span of - modifiers in the same category marked in the text. - """ - self.__SCOPEUPDATED = True - # make sure each tag has its own self-limited scope - for modifier in self.__markedModifiers: - modifier.setScope() - - # Now limit scope based on the domains of the spans of the other - # modifier - for i in range(len(self.__markedModifiers)-1): - modifier = self.__markedModifiers[i] - for j in range(i+1,len(self.__markedModifiers)): - modifier2 = self.__markedModifiers[j] - modifier.limitScope(modifier2) - modifier2.limitScope(modifier) - - def markTargets(self,terms,objType = termObject): - """tags the sentence for a list of terms - terms: a list of terms each term is a tuple with the following two elements: - term[0]--the term to tag - term[1]--the category for the tag (e.g. "EXCLUDE") - term[2]--a regular expression to express a general form of term. Pass an empty string if not desired""" - self.__markedTargets = [] - for term in terms: - self.__markedTargets.extend(self.markText(term,mode=objType)) - - def markModifiers(self,terms,objType=tagObject): - """tags the sentence for a list of terms - terms: a list of terms each term is a tuple with the following two elements: - term[0]--the term to tag - term[1]--the category for the tag (e.g. "EXCLUDE") - term[2]--a regular expression to express a general form of term. Pass an empty string if not desired - term[3]--a rule for the tag""" - self.__markedModifiers = [] - for term in terms: - self.__markedModifiers.extend(self.markText(term,mode=tagObject)) - self.__SCOPEUPDATED = False - - - - - def markText(self,term, mode=termObject, ignoreCase=True ): - """ - markup the current text with the current term. - If ignoreCase is True (default), the regular expression is compiled with - IGNORECASE. - - Note that the regular expression is only generated once. So the case sensitivity - will be determined for all subsequent uses by the first application of the term - term is a tuple with the following elements - term[0]--the text to tag - term[1]--the category - term[2]--a regular expression or empty string - term[3]--a rule (if mode='Modifier')""" - - if( not self.__txt ): - self.getCleanTxt() - - # See if we have already created a regular expression - - if(not self.res.has_key(term[0]) ): - if(not term[2]): - regExp = term[0] - else: - regExp = term[2] - r = re.compile(regExp, re.IGNORECASE) - self.res[term[0]] = r - else: - r = self.res[term[0]] - iter = r.finditer(self.__txt) - terms=[] - for i in iter: - tO = mode(term[0], term[1],term[2],term[3],self.__scope) - - tO.setStart(i.start()) - tO.setSpan(i.span()) - tO.setPhrase(i.group()) - terms.append(tO) - return terms - - def pruneMarks(self): - """ - prune Marked objects by deleting any objects that lie within the span of - another object. Currently modifiers and targets are treated separately - """ - # how to deal with the modifies and modifiedBy lists? - self.__prune_marks(self.__markedModifiers) - self.__prune_marks(self.__markedTargets) - def __prune_marks(self, marks): - marks.sort() - for t1 in marks: - for t2 in marks: - if( not t1 is t2 ): - if( t1.encompasses(t2) ): - # Need to add an adjustment of the - # targets modified by t2 - marks.remove(t2) - elif( t2.encompasses(t1) ): - marks.remove(t1) - break - def dropMarks(self,category="exclusion"): - """Drop any targets that have the category equal to category""" - # how to deal with the modifies and modifiedBy lists? - for obj in self.__markedTargets: - if( category.lower() in obj.getCategory() ): - self.__markedTargets.remove(obj) - for obj in self.__markedModifiers: - if( obj.getCategory() == category ): - self.__markedModifiers.remove(obj) - - def applyModifiers(self): - """ - If the scope has not yet been updated, do this first. - - Loop through the marked targets and for each target apply the modifiers - """ - # make sure markedModifiers is sorted - if( not self.__SCOPEUPDATED ): - self.updateScopes() - for term in self.__markedTargets: - for modifier in self.__markedModifiers: - modifier.applyRule(term) - def getMarkedTargets(self): - """ - Return the list of marked targets in the current sentence - """ - return self.__markedTargets # should I be returning a copy instead? - def getNumMarkedTargets(self): - """ - Return the number of marked targets in the current sentence - """ - return len(self.__markedTargets) - def getMarkedTarget(self,i=0): - """ - return the ith marked target - """ - - return self.__markedTargets[i] - - - - diff --git a/pyConTextNLP/deprecated/pycontextNX.py b/pyConTextNLP/deprecated/pycontextNX.py deleted file mode 100644 index 7c41bdf..0000000 --- a/pyConTextNLP/deprecated/pycontextNX.py +++ /dev/null @@ -1,59 +0,0 @@ -#Copyright 2010 Brian E. Chapman -# -#Licensed under the Apache License, Version 2.0 (the "License"); -#you may not use this file except in compliance with the License. -#You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -#Unless required by applicable law or agreed to in writing, software -#distributed under the License is distributed on an "AS IS" BASIS, -#WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -#See the License for the specific language governing permissions and -#limitations under the License. -import pycontext -import networkx as nx - -class pyConTextGraph(object): - def __init__(self, context): - self.co = context - self.graphs = {} - self.__populate() - - def __populate(self): - self.createRelationGraph() - self.createAllTagsGraph() - - def createRelationGraph(self): - self.graphs["relations"] = nx.DiGraph() - root = self.co.getText() - - for j in range(self.co.getNumMarkedTargets()): - - term = self.co.getMarkedTarget(j) - self.graphs["relations"].add_edge(root,term.getBriefDescription()) - modifiers = term.getModifiers() - for m in modifiers: - self.graphs["relations"].add_edge(term.getBriefDescription(),m.getBriefDescription()) - def createAllTagsGraph(self): - self.graphs["alltags"] = nx.DiGraph() - root = self.co.getText() - for j in range(self.co.getNumMarkedTargets()): - term = self.co.getMarkedTarget(j) - self.graphs["alltags"].add_edge(root,term.getBriefDescription()) - for j in range(self.co.getNumMarkedModifiers()): - term = self.co.getMarkedModifier(j) - self.graphs["alltags"].add_edge(root,term.getBriefDescription()) - - def update(self, graph = None): - self.createRelationGraph() - self.createAllTagsGraph() - def getGraph(self, key): - return self.graphs.get(key) - def drawGraph(self, key=None, filename="graph", format="pdf"): - graph = self.graphs.get(key) - if( not graph ): - return - if( graph.number_of_nodes() > 1 ): - ag = nx.to_pydot(graph) - ag.write("%s.%s"%(filename,format),format=format) \ No newline at end of file diff --git a/pyConTextNLP/deprecated/pycontextSql.py b/pyConTextNLP/deprecated/pycontextSql.py deleted file mode 100644 index ec71c61..0000000 --- a/pyConTextNLP/deprecated/pycontextSql.py +++ /dev/null @@ -1,132 +0,0 @@ -#Copyright 2010 Brian E. Chapman -# -#Licensed under the Apache License, Version 2.0 (the "License"); -#you may not use this file except in compliance with the License. -#You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -#Unless required by applicable law or agreed to in writing, software -#distributed under the License is distributed on an "AS IS" BASIS, -#WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -#See the License for the specific language governing permissions and -#limitations under the License. -""" -This module defines a class used for logical querying over a collection of sentences. -For the lack of knowing a better way of doing this, the sentences are dumped into a -sqlite database. -""" -import sqlite3 as sqlite -#import pyContext.pycontext as pycontext -import pyConTextGraph as pycontext -class pyConTextSql(object): - def __init__(self, db = None ): - """ - db is an optional filename for the database. If specified, a databse with the - specfied filename is generated. - - Note: if an actual filename is specified, it is assumed that a corresponding - databse does not exist. An exception will be thrown if the database exists - and the object attempts to create tabels - """ - if( db == None ): - self.memory = True - self.__conn = sqlite.connect(":memory:") - else: - self.memory = False - self.__conn = sqlite.connect(db) - self.__cursor = self.__conn.cursor() - - # create the table for sentences - self.__cursor.execute("""CREATE TABLE sentences ( - id INT PRIMARY KEY, - num INT NOT NULL, - txt text)""") - self.__cursor.execute("""CREATE TABLE foundTargets ( - id INTEGER PRIMARY KEY, - sentence INTEGER NOT NULL REFERENCES "sentences" ("id"), - term TEXT, - unmodified VARCHAR(3) DEFAULT "NO")""") - - def populate(self,context,modifierFilters): - """ - Given a context object, and columns to the foundTargets table corresponding to the - modifiers listed in modiferFilters - """ - for modfilter in modifierFilters: - self.__cursor.execute("""ALTER TABLE foundTargets ADD COLUMN %s varchar(3) DEFAULT "NO" """%"'%s'"%modfilter) - self.__cursor.execute("""ALTER TABLE foundTargets ADD COLUMN %s TEXT DEFAULT ''"""%"'%s TERM'"%modfilter) - self.__cursor.execute("""ALTER TABLE foundTargets ADD COLUMN %s TEXT DEFAULT ''"""%"'%s PHRASE'"%modfilter) - - # insert sentences into sentence table - for i in range(context.getNumberSentences()): - context.setSentence(i) - self.__cursor.execute("""INSERT INTO sentences (num,txt) VALUES (?,?)""", - (i,context.getText(),)) - self.__cursor.execute("select last_insert_rowid() from sentences") - sentid = i # self.__cursor.fetchone()[0] - for tag in context.getConTextModeNodes("target"): - - query = [] - values = [] - for modfilter in modifierFilters: - modifier = context.isModifiedBy(tag, modfilter) - if( modifier ): - query.extend(["'%s'"%modfilter,"'%s TERM'"%modfilter]) - values.extend(['YES',modifier.__str__()]) - if( not query ): - self.__cursor.execute("""INSERT INTO foundTargets (sentence,term,unmodified) VALUES (?,?,?)""", - (sentid,tag.getLiteral(),'YES',)) - else: - values.insert(0,sentid) - values.insert(1,tag.__str__()) - v = ','.join(query) - p = ','.join(["?"]*len(query)) - q = """INSERT INTO foundTargets (sentence,term,%s) VALUES (?,?,%s)"""%(v,p) - self.__cursor.execute(q,values) - if( not self.memory ): - self.__conn.commit() - - - def query(self,query): - """ - Submit query to the object to answer logical questions about the ConText markup. - Currently the query is a SQL query fragment (the user should not provide the 'SELECT * from TABLENAME' - to the foundTargets table database. - s""" - q = """SELECT * from foundTargets %s"""%query - self.__cursor.execute(q) - rslts = self.__cursor.fetchall() - sentences = [] - data = [] - if( rslts ): - for r in rslts: - self.__cursor.execute("SELECT txt from sentences where num == %d"%r[1]) - data.append((r[1:],self.__cursor.fetchone())) - return data - - - def __str__(self): - try: - txt = '' - - self.__cursor.execute("select num,txt from sentences") - sentences = self.__cursor.fetchall() - for s in sentences: - txt += "[%d]: %s\n"%(s[0],s[1]) - - txt += "\n" - self.__cursor.execute("PRAGMA table_info(foundTargets)") - tInfo = self.__cursor.fetchall() - - self.__cursor.execute("""select * from foundTargets""") - d = self.__cursor.fetchall() - for dd in d: - for i in range(len(tInfo)): - ddd = dd[i] - txt += "%s <<<%s>>> ||| "%(tInfo[i][1],ddd) - txt += "\n" - return txt - except Exception, error: - print "failed in __str__",error - return txt diff --git a/pyConTextNLP/functional/ConTextItem.py b/pyConTextNLP/functional/ConTextItem.py index d41b0e8..e7013a6 100644 --- a/pyConTextNLP/functional/ConTextItem.py +++ b/pyConTextNLP/functional/ConTextItem.py @@ -14,8 +14,14 @@ """ ConTextItem---DOCSTRING """ -import platform import collections +import csv +import urllib.request +import urllib.error +import urllib.parse +from io import StringIO +import yaml + class ConTextItem(collections.namedtuple('ConTextItem', @@ -28,6 +34,9 @@ def __str__(self): self.rule) +# def __new__(ConTextItem,l,c,rgx, rl, *args, **kwargs): +# _l = l.lower().strip() +# return super().__new__(ConTextItem, _l, c, _assign_regex(_l, rgx), rl) def _get_categories(cats): if "," in cats: return tuple([c.lower().strip() for c in cats.split(",")]) @@ -75,79 +84,30 @@ def isA(citem, testCategory): def ConTextItem2string(ci): return ci.__unicode__() -if platform.python_version_tuple()[0] == '2': - - import unicodecsv as csv - import urllib2 - - def get_fileobj(csvFile): - p = urllib2.urlparse.urlparse(csvFile) - if not p.scheme: - csvFile = "file://"+csvFile - f0 = urllib2.urlopen(csvFile, 'rU') - return csv.reader(f0, encoding='utf-8', delimiter="\t"), f0 - -else: - import csv - import urllib.request - import urllib.error - import urllib.parse - from io import StringIO - - def get_fileobj(csvFile): - p = urllib.parse.urlparse(csvFile) - if not p.scheme: - csvFile = "file://"+csvFile - f0 = urllib.request.urlopen(csvFile, data=None) - return csv.reader(StringIO(f0.read().decode(), newline=None), delimiter="\t" ), f0 - -def readConTextItems(csvFile, - encoding='utf-8', - headerRows=1, - literalColumn=0, - categoryColumn=1, - regexColumn=2, - ruleColumn=3): - """ - takes a CSV file of itemdata rules and creates a list of - ConTextItem instances. - csvFile: name of file to read items from - encoding: unicode enocidng to use; default = 'utf-8' - headerRows: number of header rows in file; default = 1 - literalColumn: column from which to read the literal; default = 0 - categoryColumn: column from which to read the category; default = 1 - regexColumn: column from which to read the regular expression: default = 2 - ruleColumn: column from which to read the rule; default = 3 - """ - items = [] - header = [] - reader, f0 = get_fileobj(csvFile) - # reader = csv.reader(open(csvFile, 'rU')) - # first grab number of specified header rows - for i in range(headerRows): - row = next(reader) - header.append(row) - # now grab each itemData - for row in reader: - tmp = [row[literalColumn], row[categoryColumn], - row[regexColumn], row[ruleColumn]] - tmp[2] = r"{0}".format(tmp[2]) - # convert the regular expression string into a raw StringIO - item = create_ConTextItem(tmp) - items.append(item) + +def get_fileobj(csvFile): + p = urllib.parse.urlparse(csvFile) + if not p.scheme: + csvFile = "file://"+csvFile + f0 = urllib.request.urlopen(csvFile, data=None) + return csv.reader(StringIO(f0.read().decode(), newline=None), delimiter="\t" ), f0 + + + + +def _get_fileobj(_file): + if not urllib.parse.urlparse(_file).scheme: + _file = "file://"+_file + return urllib.request.urlopen(_file, data=None) + +def get_items(_file): + f0 = _get_fileobj(_file) + context_items = [ConTextItem(literal=d["Lex"], + category=d["Type"], + re=r"%s"%d["Regex"], + rule=d["Direction"]) for d in yaml.load_all(f0)] f0.close() - return items, header + return context_items + -def writeConTextItems(items, fname): - """ - Write the ConTextItems as a tab delimited file to the file specified - in fname - """ - with open(fname, 'w') as f0: - f0.write("Lex\tType\tRegex\tRule\n") - for i in items: - f0.write("%s\t%s\t%s\t%s\n" % (i.literal, - ",".join(i.category), - i.re, - i.rule)) diff --git a/pyConTextNLP/functional/tests/run_tests.sh b/pyConTextNLP/functional/tests/run_tests.sh deleted file mode 100644 index cf8fca1..0000000 --- a/pyConTextNLP/functional/tests/run_tests.sh +++ /dev/null @@ -1 +0,0 @@ -nose2 -t ../../../pyConTextNLP diff --git a/pyConTextNLP/functional/tests/setup.cfg b/pyConTextNLP/functional/tests/setup.cfg deleted file mode 100644 index b25544b..0000000 --- a/pyConTextNLP/functional/tests/setup.cfg +++ /dev/null @@ -1,4 +0,0 @@ -[nosy] -# Paths to check for changed files; changes cause nose to be run -base_path = ./ -glob_patterns = *.py diff --git a/pyConTextNLP/functional/tests/test_contextitem.py b/pyConTextNLP/functional/tests/test_contextitem.py new file mode 100644 index 0000000..c189280 --- /dev/null +++ b/pyConTextNLP/functional/tests/test_contextitem.py @@ -0,0 +1,60 @@ +import pyConTextNLP.functional.ConTextItem as CI +import pytest + +@pytest.fixture(scope="module") +def items(): + + return [ ["pulmonary embolism", + ["PULMONARY_EMBOLISM"], + r"""pulmonary\s(artery )?(embol[a-z]+)""", + ""], + ["no gross evidence of", + [ "PROBABLE_NEGATED_EXISTENCE"], + "", + "forward"]] + +def test_instantiate_ConTextItem0(items): + for item in items: + assert CI.ConTextItem(*item) + + +def test_ConTextItem_rule(items): + cti = CI.ConTextItem(*(items[1])) + + assert cti.rule == "forward" + + +def test_ConTextItem_literal(items): + cti = CI.ConTextItem(*(items[0])) + + assert cti.literal == "pulmonary embolism" + + +def test_ConTextItem_category(items): + cti = CI.ConTextItem(*(items[1])) + assert cti.category == ["probable_negated_existence"] + +def test_ConTextItem_isa(items): + cti = CI.ConTextItem(*(items[0])) + assert CI.isA(cti, "pulmonary_embolism") + + +def test_ConTextItem_isa1(items): + cti = CI.ConTextItem(*(items[0])) + assert CI.isA(cti, "PULMONARY_EMBOLISM") + + +def test_ConTextItem_isa2(items): + cti = CI.ConTextItem(*(items[1])) + assert CI.isA(cti, "PROBABLE_NEGATED_EXISTENCE") + + +def test_ConTextItem_getRE(items): + cti = CI.ConTextItem(*(items[1])) + assert cti.re == r'\b%s\b'%items[1][0] + + +def test_ConTextItem_getRE1(items): + cti = CI.ConTextItem(*(items[0])) + assert cti.re == r"""pulmonary\s(artery )?(embol[a-z]+)""" + diff --git a/pyConTextNLP/functional/tests/test_contextmarkup.py b/pyConTextNLP/functional/tests/test_contextmarkup.py new file mode 100644 index 0000000..546da2f --- /dev/null +++ b/pyConTextNLP/functional/tests/test_contextmarkup.py @@ -0,0 +1,45 @@ +from pyConTextNLP.ConTextMarkup import ConTextMarkup +import pytest + +@pytest.fixture(scope="module") +def sent1(): + return 'kanso **diabetes** utesl\xf6t eller diabetes men inte s\xe4kert. Vi siktar p\xe5 en r\xf6ntgenkontroll. kan det vara nej panik\xe5ngesten\n?' + +@pytest.fixture(scope="module") +def sent2(): + return 'IMPRESSION: 1. LIMITED STUDY DEMONSTRATING NO GROSS EVIDENCE OF SIGNIFICANT PULMONARY EMBOLISM.' +@pytest.fixture(scope="module") +def sent3(): + return 'This is a sentence that does not end with a number. But this sentence ends with 1. So this should be recognized as a third sentence.' + +@pytest.fixture(scope="module") +def sent4(): + return 'This is a sentence with a numeric value equal to 1.43 and should not be split into two parts.' + +@pytest.fixture(scope="module") +def items(): + return [ ["pulmonary embolism", + "PULMONARY_EMBOLISM", + r"""pulmonary\s(artery )?(embol[a-z]+)""", + ""], + ["no gross evidence of", + "PROBABLE_NEGATED_EXISTENCE", + "", + "forward"]] + +def test_setRawText1(sent1): + context = ConTextMarkup() + context.setRawText(sent1) + assert context.getRawText() == sent1 + +def test_scrub_preserve_unicode(sent1): + context = ConTextMarkup() + context.setRawText(sent1) + context.cleanText(stripNonAlphaNumeric=True) + assert context.getText().index(u'\xf6') == 40 + +def test_scrub_text(sent2): + context = ConTextMarkup() + context.setRawText(sent2) + context.cleanText(stripNonAlphaNumeric=True) + assert context.getText().rfind(u'.') == -1 diff --git a/pyConTextNLP/functional/tests/test_env.py b/pyConTextNLP/functional/tests/test_env.py new file mode 100644 index 0000000..c73c9b0 --- /dev/null +++ b/pyConTextNLP/functional/tests/test_env.py @@ -0,0 +1,11 @@ +def test_yaml(): + import yaml + assert yaml + +def test_networkx(): + import networkx as nx + assert nx + +def test_networkx_v2x(): + import networkx as nx + assert nx.__version__[0] == '2' diff --git a/pyConTextNLP/functional/tests/test_helpers.py b/pyConTextNLP/functional/tests/test_helpers.py new file mode 100644 index 0000000..f6fba11 --- /dev/null +++ b/pyConTextNLP/functional/tests/test_helpers.py @@ -0,0 +1,31 @@ +import pyConTextNLP.helpers as helpers +import pytest + +@pytest.fixture(scope="module") +def splitter(): + return helpers.sentenceSplitter() + + +def test_createSentenceSplitter(): + assert helpers.sentenceSplitter() + + +def test_getExceptionTerms(splitter): + assert splitter.getExceptionTerms() + + +def test_addExceptionTermsWithoutCaseVariants(splitter): + splitter.addExceptionTerms("D.D.S.", "D.O.") + assert ("D.O." in splitter.getExceptionTerms()) + #assert ("d.o." in splitter.getExceptionTerms()) + + +def test_addExceptionTermsWithCaseVariants(splitter): + splitter.addExceptionTerms("D.D.S.", "D.O.",addCaseVariants=True) + assert ("d.o." in splitter.getExceptionTerms()) + + +def test_deleteExceptionTermsWithoutCaseVariants(splitter): + splitter.deleteExceptionTerms("M.D.") + assert ("M.D." not in splitter.getExceptionTerms()) + assert ("m.d." in splitter.getExceptionTerms()) diff --git a/pyConTextNLP/functional/tests/test_itemData.py b/pyConTextNLP/functional/tests/test_itemData.py new file mode 100644 index 0000000..8454faa --- /dev/null +++ b/pyConTextNLP/functional/tests/test_itemData.py @@ -0,0 +1,25 @@ +import pyConTextNLP.functional.ConTextItem as CI +from pathlib import PurePath +import os +import pytest + + +@pytest.fixture(scope="session") +def get_tmp_dirs(): + pass + +def test_get_fileobj_1(): + fobj = PurePath(PurePath(os.path.abspath(__file__)).parent, "..", "..", "KB", "test.yml") + yaml_fo = CI.get_fileobj(str(fobj)) + assert yaml_fo + +def test_get_fileobj_2(): + wdir = PurePath(os.path.abspath(__file__))#, "..", "..", "KB") + fobj = PurePath(wdir.parent, "..", "..", "KB", "test.yml") + yfo = CI.get_fileobj("file://"+str(fobj)) + assert yfo + +def test_get_fileobj_3(): + yfo = CI.get_fileobj( + "https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/test.yml") + assert yfo diff --git a/pyConTextNLP/functional/tests/tests1.py b/pyConTextNLP/functional/tests/tests1.py deleted file mode 100644 index 041c990..0000000 --- a/pyConTextNLP/functional/tests/tests1.py +++ /dev/null @@ -1,222 +0,0 @@ -import unittest -import pyConTextNLP.functional.ConTextItem as CI -import pyConTextNLP.functional.tagItem as TI -import pyConTextNLP.functional.ConTextMarkup as CM -import pyConTextNLP.functional.ConTextDocument as CD -from textblob import TextBlob -import networkx as nx - -class functional_test_ConTextItem(unittest.TestCase): - def setUp(self): - # create a sample image in memory - - self.items = {"targets":[ ["pulmonary embolism", - "PULMONARY_EMBOLISM", - r"""pulmonary\s(artery )?(embol[a-z]+)""",""]], - "modifiers":[["no gross evidence of", - "PROBABLE_NEGATED_EXISTENCE,HEDGE_TERM","","forward"]]} - self.files = {} - #options['infile'] = os.path.join(UTAHDATA,"CTPA PHI - main data set, 2015-04-19.xls") - self.files['lexical_kb'] = ["https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/lexical_kb_04292013.tsv", - "https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/criticalfinder_generalized_modifiers.tsv", - "https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/critical_modifiers.tsv"] - self.files['domain_kb'] = ["https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/utah_crit.tsv"]#[os.path.join(DATADIR2,"pe_kb.tsv")] - - - def tearDown(self): - self.items = 0 - def test_create_ConTextItem(self): - ci = CI.create_ConTextItem(self.items['targets'][0]) - assert ci.re == self.items['targets'][0][2] - assert ci.category == (self.items['targets'][0][1].lower().strip(),) - - def test_create_ConTextItem2(self): - ci = CI.create_ConTextItem(self.items['modifiers'][0]) - assert ci.re == self.items['modifiers'][0][0].lower().strip() - assert len(ci.category) == 2 - - def test_isA(self): - ci = CI.create_ConTextItem(self.items["modifiers"][0]) - assert CI.isA(ci,"HEART_DISEASE") == False - assert CI.isA(ci,"HEDGE_TERM") == True - def test_test_rule(self): - ci = CI.create_ConTextItem(self.items["modifiers"][0]) - assert CI.test_rule(ci,"Forward") == True - assert CI.test_rule(ci,"forward") == True - assert CI.test_rule(ci,"bidirection") == False - - def test_read_ConTextItem0(self): - items, headers = CI.readConTextItems(self.files['lexical_kb'][0]) - assert items - - def test_read_ConTextItem1(self): - items, headers = CI.readConTextItems(self.files['lexical_kb'][1]) - assert items - -class functional_test_tagItem(unittest.TestCase): - def setUp(self): - # create a sample image in memory - - self.items = {"targets":[ CI.create_ConTextItem([u"pulmonary embolism", - u"PULMONARY_EMBOLISM", - r"""pulmonary\s(artery )?(embol[a-z]+)""",""])], - "modifiers":[CI.create_ConTextItem(["no gross evidence of", - "PROBABLE_NEGATED_EXISTENCE,HEDGE_TERM","","forward"])]} - self.ci0 = self.items['targets'][0] - self.ci1 = self.items['modifiers'][0] - - def tearDown(self): - self.items = 0 - self.ci0 = 0 - self.ci1 = 0 - def test_create_tagItem(self): - ti = TI.create_tagItem(self.ci0,(20,30),"Brian Chapman",0) - assert ti.span == (20,30) - def test_limitCategoryScopeBackward(self): - ti0 = TI.create_tagItem(self.ci0,(20,30),"Brian Chapman",0) - ti1 = TI.create_tagItem(self.ci0,( 5,15),"Wendy Chapman",1) - assert TI.lessthan(ti1,ti0) - assert not TI.lessthan(ti0,ti1) - def test_o1_encompasses_o2(self): - - ti0 = TI.create_tagItem(self.ci0,(20,25),"Brian Chapman",0) - ti1 = TI.create_tagItem(self.ci0,( 5,30),"Wendy Chapman",1) - assert TI.o1_encompasses_o2(ti1,ti0) - assert not TI.o1_encompasses_o2(ti0,ti1) - - # add function to test DocumentGraph generation - - -class functional_test_ConTextMarkup(unittest.TestCase): - def setUp(self): - # create a sample image in memory - - - self.su1 = u'kanso **diabetes** utesl\xf6t eller diabetes men inte s\xe4kert. Vi siktar p\xe5 en r\xf6ntgenkontroll. kan det vara nej panik\xe5ngesten\n?' - self.su2 = u'IMPRESSION: 1. LIMITED STUDY DEMONSTRATING NO GROSS EVIDENCE OF SIGNIFICANT PULMONARY EMBOLISM.' - self.su3 = u'This is a sentence that does not end with a number. But this sentence ends with 1. So this should be recognized as a third sentence.' - self.su4 = u'This is a sentence with a numeric value equal to 1.43 and should not be split into two parts.' - self.items = {"targets":[ CI.create_ConTextItem([u"pulmonary embolism", - u"PULMONARY_EMBOLISM", - r"""pulmonary\s(artery )?(embol[a-z]+)""", - ""]), - ], - "modifiers":[CI.create_ConTextItem(["no gross evidence of", - "PROBABLE_NEGATED_EXISTENCE", - "", - "forward"])]} - - - - self.ci0 = self.items['targets'][0] - self.ci1 = self.items['modifiers'][0] - self.ti0 = TI.create_tagItem(self.ci0,(20,25),"Brian Chapman",0) - self.ti1 = TI.create_tagItem(self.ci0,( 5,30),"Wendy Chapman",1) - self.markup = nx.DiGraph() - self.markup.add_nodes_from([self.ti0,self.ti1]) - - - def tearDown(self): - self.su1 = 0 - self.su2 = 0 - self.su3 = 0 - self.su4 = 0 - self.items = 0 - self.ci0 = 0 - self.ci1 = 0 - self.ti0 = 0 - self.ti1 = 0 - self.markup = 0 - def test_create_markup(self): - assert self.markup - - def test_setRawText(self): - self.markup = CM.setRawText(self.markup, self.su1) - assert self.markup.graph["__rawText"] == self.su1 - def test_scrub_preserve_unicode(self): - self.markup = CM.setRawText(self.markup, self.su1) - self.markup = CM.cleanText(self.markup, stripNonAlphaNumeric=True) - assert self.markup.graph["__text"].index(u'\xf6') == 40 - def test_scrub_text(self): - self.markup = CM.setRawText(self.markup, self.su2) - self.markup = CM.cleanText( self.markup, stripNonAlphaNumeric=True) - assert self.markup.graph["__text"].rfind(u'.') == -1 - - def test_prune_marks(self): - markup_new = CM.prune_marks(self.markup) - assert len(markup_new) == 1 - assert markup_new.nodes()[0].id == 1 - def test_mark_sentence(self): - mu = CM.mark_sentence(self.su2,self.items) - assert mu - -class functional_test_ConTextDocument(unittest.TestCase): - def setUp(self): - # create a sample image in memory - - - self.su1 = u'kanso **diabetes** utesl\xf6t eller diabetes men inte s\xe4kert. Vi siktar p\xe5 en r\xf6ntgenkontroll. kan det vara nej panik\xe5ngesten\n?' - self.su2 = u'IMPRESSION: 1. LIMITED STUDY DEMONSTRATING NO GROSS EVIDENCE OF SIGNIFICANT PULMONARY EMBOLISM.' - self.su3 = u'This is a sentence that does not end with a number. But this sentence ends with 1. So this should be recognized as a third sentence.' - self.su4 = u'This is a sentence with a numeric value equal to 1.43 and should not be split into two parts.' - self.items = {"targets":[ CI.create_ConTextItem([u"pulmonary embolism", - u"PULMONARY_EMBOLISM", - r"""pulmonary\s(artery )?(embol[a-z]+)""",""])], - "modifiers":[CI.create_ConTextItem(["no gross evidence of", - "PROBABLE_NEGATED_EXISTENCE","","forward"])]} - - - - self.ci0 = self.items['targets'][0] - self.ci1 = self.items['modifiers'][0] - self.ti0 = TI.create_tagItem(self.ci0,(20,25),"Brian Chapman",0) - self.ti1 = TI.create_tagItem(self.ci0,( 5,30),"Wendy Chapman",1) - self.markup = nx.DiGraph() - self.markup.add_nodes_from([self.ti0,self.ti1]) - self.cd = CD.ConTextDocument() - - - - def tearDown(self): - self.su1 = 0 - self.su2 = 0 - self.su3 = 0 - self.su4 = 0 - self.items = 0 - self.ci0 = 0 - self.ci1 = 0 - self.ti0 = 0 - self.ti1 = 0 - self.markup = 0 - self.cd = 0 - -# - - - def test_create_ConTextDocument(self): - cd = CD.ConTextDocument() - assert cd.graph["__documentGraph"] == None - - def test_TextBlob1(self): - blob = TextBlob(self.su1) - assert len(blob.sentences) == 3 - - - def test_TextBlob3(self): - blob = TextBlob(self.su3) - assert len(blob.sentences) == 3 - - def test_TextBlob4(self): - blob = TextBlob(self.su4) - assert len(blob.sentences) == 1 - - def test_insertSection(self): - cd2 = CD.insertSection(self.cd, - "report", - setToParent=True, - setToRoot = True) - assert CD.getCurrentparent(cd2) == "report" - - -def run(): - pass diff --git a/pyConTextNLP/tests/test1.py b/pyConTextNLP/tests/test1.py deleted file mode 100644 index 33e5b97..0000000 --- a/pyConTextNLP/tests/test1.py +++ /dev/null @@ -1,91 +0,0 @@ -import unittest -import pyConTextNLP.itemData as itemData -import pyConTextNLP.pyConTextGraph as pyConText -import pyConTextNLP.helpers as helpers -import pyConTextNLP.itemData as itemData -import os -from textblob import TextBlob - -class pyConTextNLP_test(unittest.TestCase): - def setUp(self): - # create a sample image in memory - self.context = pyConText.ConTextMarkup() - self.splitter = helpers.sentenceSplitter() - - - self.su1 = u'kanso **diabetes** utesl\xf6t eller diabetes men inte s\xe4kert. Vi siktar p\xe5 en r\xf6ntgenkontroll. kan det vara nej panik\xe5ngesten\n?' - self.su2 = u'IMPRESSION: 1. LIMITED STUDY DEMONSTRATING NO GROSS EVIDENCE OF SIGNIFICANT PULMONARY EMBOLISM.' - self.su3 = u'This is a sentence that does not end with a number. But this sentence ends with 1. So this should be recognized as a third sentence.' - self.su4 = u'This is a sentence with a numeric value equal to 1.43 and should not be split into two parts.' - self.items = [ [u"pulmonary embolism",u"PULMONARY_EMBOLISM",ur"""pulmonary\s(artery )?(embol[a-z]+)""",""],["no gross evidence of","PROBABLE_NEGATED_EXISTENCE","","forward"]] - self.itemData = itemData.itemData() - for i in self.items: - cit = itemData.contextItem - - def tearDown(self): - self.context = 0 - self.splitter = 0 - self.su1 = 0 - #def testSource(self): - #assert self.context.__file__ == 'pyConTextGraph.pyc' - def test_itemData_from_tsv(self): - f = "https://github.com/chapmanbe/pyConTextNLP/blob/master/KB/domain_kb_test.tsv" - assert True - #assert itemData.itemData_from_tsv(f) - def test_itemData_from_tsv2(self): - f = "file://"+os.path.join(os.getcwd(),"../KB/criticalfinder_generalized_modifiers.tsv") - assert itemData.itemData_from_tsv(f) - def test_setRawText(self): - self.context.setRawText(self.su1) - assert self.context.getRawText() == self.su1 - def test_scrub_preserve_unicode(self): - self.context.setRawText(self.su1) - self.context.cleanText(stripNonAlphaNumeric=True) - assert self.context.getText().index(u'\xf6') == 40 - def test_scrub_text(self): - self.context.setRawText(self.su2) - self.context.cleanText(stripNonAlphaNumeric=True) - assert self.context.getText().rfind(u'.') == -1 - def test_createSentenceSplitter(self): - assert helpers.sentenceSplitter() - def test_getExceptionTerms(self): - assert self.splitter.getExceptionTerms() - def test_addExceptionTermsWithoutCaseVariants(self): - self.splitter.addExceptionTerms("D.D.S.", "D.O.") - assert ("D.O." in self.splitter.getExceptionTerms()) - assert ("d.o." not in self.splitter.getExceptionTerms()) - def test_addExceptionTermsWithCaseVariants(self): - self.splitter.addExceptionTerms("D.D.S.", "D.O.",addCaseVariants=True) - assert ("d.o." in self.splitter.getExceptionTerms()) - def test_deleteExceptionTermsWithoutCaseVariants(self): - self.splitter.deleteExceptionTerms("M.D.") - assert ("M.D." not in self.splitter.getExceptionTerms()) - assert ("m.d." in self.splitter.getExceptionTerms()) - def test_instantiate_contextItem(self): - cit1 = itemData.contextItem(self.items[0]) - assert cit1 - def test_instantiate_itemData(self): - cit1 = itemData.contextItem(self.items[0]) - it1 = itemData.itemData() - it1.append(cit1) - assert it1 - #def test_tokenDistance(self): - #assert False - def test_sentenceSplitter1(self): - """test whether we properly capture text that terminates without a recognized sentence termination""" - splitter = helpers.sentenceSplitter() - sentences = splitter.splitSentences(self.su3) - assert len(sentences) == 3 - def test_sentenceSplitter2(self): - """test whether we properly skip numbers with decimal points.""" - splitter = helpers.sentenceSplitter() - sentences = splitter.splitSentences(self.su4) - assert len(sentences) == 1 - def test_TextBlob_sentenceSplitter(self): - blob = TextBlob(self.su3) - assert len(blob.sentences) == 3 - - # add function to test DocumentGraph generation -def run(): - pass - diff --git a/pyConTextNLP/tests/test_contextitem.py b/pyConTextNLP/tests/test_contextitem.py new file mode 100644 index 0000000..890b56c --- /dev/null +++ b/pyConTextNLP/tests/test_contextitem.py @@ -0,0 +1,60 @@ +import pyConTextNLP.itemData as itemData +import pytest + +@pytest.fixture(scope="module") +def items(): + + return [ ["pulmonary embolism", + "PULMONARY_EMBOLISM", + r"""pulmonary\s(artery )?(embol[a-z]+)""", + ""], + ["no gross evidence of", + "PROBABLE_NEGATED_EXISTENCE", + "", + "forward"]] + +def test_instantiate_contextItem0(items): + for item in items: + assert itemData.contextItem(item) + + +def test_contextItem_rule(items): + cti = itemData.contextItem(items[1]) + + assert cti.getRule() == "forward" + + +def test_contextItem_literal(items): + cti = itemData.contextItem(items[0]) + + assert cti.getLiteral() == "pulmonary embolism" + + +def test_contextItem_category(items): + cti = itemData.contextItem(items[1]) + assert cti.getCategory() == ["probable_negated_existence"] + +def test_contextItem_isa(items): + cti = itemData.contextItem(items[0]) + assert cti.isA("pulmonary_embolism") + + +def test_contextItem_isa1(items): + cti = itemData.contextItem(items[0]) + assert cti.isA("PULMONARY_EMBOLISM") + + +def test_contextItem_isa2(items): + cti = itemData.contextItem(items[1]) + assert cti.isA("PROBABLE_NEGATED_EXISTENCE") + + +def test_contextItem_getRE(items): + cti = itemData.contextItem(items[1]) + assert cti.getRE() == r'\b%s\b'%items[1][0] + + +def test_contextItem_getRE1(items): + cti = itemData.contextItem(items[0]) + assert cti.getRE() == r"""pulmonary\s(artery )?(embol[a-z]+)""" + diff --git a/pyConTextNLP/tests/test_contextmarkup.py b/pyConTextNLP/tests/test_contextmarkup.py new file mode 100644 index 0000000..546da2f --- /dev/null +++ b/pyConTextNLP/tests/test_contextmarkup.py @@ -0,0 +1,45 @@ +from pyConTextNLP.ConTextMarkup import ConTextMarkup +import pytest + +@pytest.fixture(scope="module") +def sent1(): + return 'kanso **diabetes** utesl\xf6t eller diabetes men inte s\xe4kert. Vi siktar p\xe5 en r\xf6ntgenkontroll. kan det vara nej panik\xe5ngesten\n?' + +@pytest.fixture(scope="module") +def sent2(): + return 'IMPRESSION: 1. LIMITED STUDY DEMONSTRATING NO GROSS EVIDENCE OF SIGNIFICANT PULMONARY EMBOLISM.' +@pytest.fixture(scope="module") +def sent3(): + return 'This is a sentence that does not end with a number. But this sentence ends with 1. So this should be recognized as a third sentence.' + +@pytest.fixture(scope="module") +def sent4(): + return 'This is a sentence with a numeric value equal to 1.43 and should not be split into two parts.' + +@pytest.fixture(scope="module") +def items(): + return [ ["pulmonary embolism", + "PULMONARY_EMBOLISM", + r"""pulmonary\s(artery )?(embol[a-z]+)""", + ""], + ["no gross evidence of", + "PROBABLE_NEGATED_EXISTENCE", + "", + "forward"]] + +def test_setRawText1(sent1): + context = ConTextMarkup() + context.setRawText(sent1) + assert context.getRawText() == sent1 + +def test_scrub_preserve_unicode(sent1): + context = ConTextMarkup() + context.setRawText(sent1) + context.cleanText(stripNonAlphaNumeric=True) + assert context.getText().index(u'\xf6') == 40 + +def test_scrub_text(sent2): + context = ConTextMarkup() + context.setRawText(sent2) + context.cleanText(stripNonAlphaNumeric=True) + assert context.getText().rfind(u'.') == -1 diff --git a/pyConTextNLP/tests/test_env.py b/pyConTextNLP/tests/test_env.py new file mode 100644 index 0000000..c73c9b0 --- /dev/null +++ b/pyConTextNLP/tests/test_env.py @@ -0,0 +1,11 @@ +def test_yaml(): + import yaml + assert yaml + +def test_networkx(): + import networkx as nx + assert nx + +def test_networkx_v2x(): + import networkx as nx + assert nx.__version__[0] == '2' diff --git a/pyConTextNLP/tests/test_helpers.py b/pyConTextNLP/tests/test_helpers.py new file mode 100644 index 0000000..f6fba11 --- /dev/null +++ b/pyConTextNLP/tests/test_helpers.py @@ -0,0 +1,31 @@ +import pyConTextNLP.helpers as helpers +import pytest + +@pytest.fixture(scope="module") +def splitter(): + return helpers.sentenceSplitter() + + +def test_createSentenceSplitter(): + assert helpers.sentenceSplitter() + + +def test_getExceptionTerms(splitter): + assert splitter.getExceptionTerms() + + +def test_addExceptionTermsWithoutCaseVariants(splitter): + splitter.addExceptionTerms("D.D.S.", "D.O.") + assert ("D.O." in splitter.getExceptionTerms()) + #assert ("d.o." in splitter.getExceptionTerms()) + + +def test_addExceptionTermsWithCaseVariants(splitter): + splitter.addExceptionTerms("D.D.S.", "D.O.",addCaseVariants=True) + assert ("d.o." in splitter.getExceptionTerms()) + + +def test_deleteExceptionTermsWithoutCaseVariants(splitter): + splitter.deleteExceptionTerms("M.D.") + assert ("M.D." not in splitter.getExceptionTerms()) + assert ("m.d." in splitter.getExceptionTerms()) diff --git a/pyConTextNLP/tests/test_itemData.py b/pyConTextNLP/tests/test_itemData.py new file mode 100644 index 0000000..614c596 --- /dev/null +++ b/pyConTextNLP/tests/test_itemData.py @@ -0,0 +1,25 @@ +import pyConTextNLP.itemData as itemData +from pathlib import PurePath +import os +import pytest + + +@pytest.fixture(scope="session") +def get_tmp_dirs(): + pass + +def test_get_fileobj_1(): + fobj = PurePath(PurePath(os.path.abspath(__file__)).parent, "..", "..", "KB", "test.yml") + yaml_fo = itemData.get_fileobj(str(fobj)) + assert yaml_fo + +def test_get_fileobj_2(): + wdir = PurePath(os.path.abspath(__file__))#, "..", "..", "KB") + fobj = PurePath(wdir.parent, "..", "..", "KB", "test.yml") + yfo = itemData.get_fileobj("file://"+str(fobj)) + assert yfo + +def test_get_fileobj_3(): + yfo = itemData.get_fileobj( + "https://raw.githubusercontent.com/chapmanbe/pyConTextNLP/master/KB/test.yml") + assert yfo diff --git a/pyConTextNLP/version.py b/pyConTextNLP/version.py index a7bde01..b780c7c 100644 --- a/pyConTextNLP/version.py +++ b/pyConTextNLP/version.py @@ -1 +1 @@ -__version__="0.6.2.1" +__version__="0.7.0.0" diff --git a/requirements-py2.txt b/requirements-py2.txt deleted file mode 100644 index fee1540..0000000 --- a/requirements-py2.txt +++ /dev/null @@ -1 +0,0 @@ -unicodecsv diff --git a/setup.py b/setup.py index f89ff0f..71d1e10 100644 --- a/setup.py +++ b/setup.py @@ -1,7 +1,6 @@ from setuptools import setup, find_packages # Always prefer setuptools over distutils from codecs import open # To use a consistent encoding from os import path -import json version = {} with open(path.join("pyConTextNLP","version.py")) as f0: @@ -54,7 +53,7 @@ # that you indicate whether you support Python 2, Python 3 or both. #'Programming Language :: Python :: 2', #'Programming Language :: Python :: 2.6', - 'Programming Language :: Python :: 2.7', + #'Programming Language :: Python :: 2.7', 'Programming Language :: Python :: 3', ], @@ -63,13 +62,13 @@ # You can just specify the packages manually here if your project is # simple. Or you can use find_packages(). - packages=find_packages(exclude=['contrib', 'docs', 'functional','tests*']), + packages=find_packages(exclude=['contrib', 'docs', 'tests*']), # List run-time dependencies here. These will be installed by pip when your # project is installed. For an analysis of "install_requires" vs pip's # requirements files see: # https://packaging.python.org/en/latest/requirements.html - install_requires=['networkx'], + install_requires=['networkx>2.0', 'frozendict'], # List additional groups of dependencies here (e.g. development dependencies). # You can install these using the following syntax, for example: