notation3.py 66 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068
  1. """
  2. notation3.py - Standalone Notation3 Parser
  3. Derived from CWM, the Closed World Machine
  4. Authors of the original suite:
  5. * Dan Connolly <@@>
  6. * Tim Berners-Lee <@@>
  7. * Yosi Scharf <@@>
  8. * Joseph M. Reagle Jr. <reagle@w3.org>
  9. * Rich Salz <rsalz@zolera.com>
  10. http://www.w3.org/2000/10/swap/notation3.py
  11. Copyright 2000-2007, World Wide Web Consortium.
  12. Copyright 2001, MIT.
  13. Copyright 2001, Zolera Systems Inc.
  14. License: W3C Software License
  15. http://www.w3.org/Consortium/Legal/copyright-software
  16. Modified by Sean B. Palmer
  17. Copyright 2007, Sean B. Palmer.
  18. Modified to work with rdflib by Gunnar Aastrand Grimnes
  19. Copyright 2010, Gunnar A. Grimnes
  20. """
  21. from __future__ import annotations
  22. import codecs
  23. import os
  24. import re
  25. import sys
  26. # importing typing for `typing.List` because `List`` is used for something else
  27. import typing
  28. from decimal import Decimal
  29. from typing import (
  30. IO,
  31. TYPE_CHECKING,
  32. Any,
  33. Callable,
  34. Dict,
  35. Match,
  36. MutableSequence,
  37. NoReturn,
  38. Optional,
  39. Pattern,
  40. Set,
  41. Tuple,
  42. TypeVar,
  43. Union,
  44. )
  45. from uuid import uuid4
  46. from rdflib.compat import long_type
  47. from rdflib.exceptions import ParserError
  48. from rdflib.graph import Dataset, Graph, QuotedGraph
  49. from rdflib.term import (
  50. _XSD_PFX,
  51. BNode,
  52. IdentifiedNode,
  53. Identifier,
  54. Literal,
  55. Node,
  56. URIRef,
  57. Variable,
  58. _unique_id,
  59. )
  60. __all__ = [
  61. "BadSyntax",
  62. "N3Parser",
  63. "TurtleParser",
  64. "splitFragP",
  65. "join",
  66. "base",
  67. "runNamespace",
  68. "uniqueURI",
  69. "hexify",
  70. "Formula",
  71. "RDFSink",
  72. "SinkParser",
  73. "sfloat",
  74. ]
  75. from rdflib.parser import Parser
  76. if TYPE_CHECKING:
  77. from rdflib.parser import InputSource
  78. _AnyT = TypeVar("_AnyT")
  79. def splitFragP(uriref: str, punc: int = 0) -> Tuple[str, str]:
  80. """Split a URI reference before the fragment
  81. Punctuation is kept. e.g.
  82. ```python
  83. >>> splitFragP("abc#def")
  84. ('abc', '#def')
  85. >>> splitFragP("abcdef")
  86. ('abcdef', '')
  87. ```
  88. """
  89. i = uriref.rfind("#")
  90. if i >= 0:
  91. return uriref[:i], uriref[i:]
  92. else:
  93. return uriref, ""
  94. _StrT = TypeVar("_StrT", bound=str)
  95. def join(here: str, there: str) -> str:
  96. """join an absolute URI and URI reference
  97. (non-ascii characters are supported/doctested;
  98. haven't checked the details of the IRI spec though)
  99. `here` is assumed to be absolute.
  100. `there` is URI reference.
  101. ```python
  102. >>> join('http://example/x/y/z', '../abc')
  103. 'http://example/x/abc'
  104. ```
  105. Raise ValueError if there uses relative path
  106. syntax but here has no hierarchical path.
  107. ```python
  108. >>> join('mid:foo@example', '../foo') # doctest: +NORMALIZE_WHITESPACE
  109. Traceback (most recent call last):
  110. raise ValueError(here)
  111. ValueError: Base <mid:foo@example> has no slash
  112. after colon - with relative '../foo'.
  113. >>> join('http://example/x/y/z', '')
  114. 'http://example/x/y/z'
  115. >>> join('mid:foo@example', '#foo')
  116. 'mid:foo@example#foo'
  117. ```
  118. We grok IRIs
  119. ```python
  120. >>> len('Andr\\xe9')
  121. 5
  122. >>> join('http://example.org/', '#Andr\\xe9')
  123. 'http://example.org/#Andr\\xe9'
  124. ```
  125. """
  126. # assert(here.find("#") < 0), \
  127. # "Base may not contain hash: '%s'" % here # why must caller splitFrag?
  128. slashl = there.find("/")
  129. colonl = there.find(":")
  130. # join(base, 'foo:/') -- absolute
  131. if colonl >= 0 and (slashl < 0 or colonl < slashl):
  132. return there
  133. bcolonl = here.find(":")
  134. assert bcolonl >= 0, (
  135. "Base uri '%s' is not absolute" % here
  136. ) # else it's not absolute
  137. path, frag = splitFragP(there)
  138. if not path:
  139. return here + frag
  140. # join('mid:foo@example', '../foo') bzzt
  141. if here[bcolonl + 1 : bcolonl + 2] != "/":
  142. raise ValueError(
  143. "Base <%s> has no slash after "
  144. "colon - with relative '%s'." % (here, there)
  145. )
  146. if here[bcolonl + 1 : bcolonl + 3] == "//":
  147. bpath = here.find("/", bcolonl + 3)
  148. else:
  149. bpath = bcolonl + 1
  150. # join('http://xyz', 'foo')
  151. if bpath < 0:
  152. bpath = len(here)
  153. here = here + "/"
  154. # join('http://xyz/', '//abc') => 'http://abc'
  155. if there[:2] == "//":
  156. return here[: bcolonl + 1] + there
  157. # join('http://xyz/', '/abc') => 'http://xyz/abc'
  158. if there[:1] == "/":
  159. return here[:bpath] + there
  160. slashr = here.rfind("/")
  161. while 1:
  162. if path[:2] == "./":
  163. path = path[2:]
  164. if path == ".":
  165. path = ""
  166. elif path[:3] == "../" or path == "..":
  167. path = path[3:]
  168. i = here.rfind("/", bpath, slashr)
  169. if i >= 0:
  170. here = here[: i + 1]
  171. slashr = i
  172. else:
  173. break
  174. return here[: slashr + 1] + path + frag
  175. def base() -> str:
  176. """The base URI for this process - the Web equiv of cwd
  177. Relative or absolute unix-standard filenames parsed relative to
  178. this yield the URI of the file.
  179. If we had a reliable way of getting a computer name,
  180. we should put it in the hostname just to prevent ambiguity
  181. """
  182. # return "file://" + hostname + os.getcwd() + "/"
  183. return "file://" + _fixslash(os.getcwd()) + "/"
  184. def _fixslash(s: str) -> str:
  185. """Fix windowslike filename to unixlike - (#ifdef WINDOWS)"""
  186. s = s.replace("\\", "/")
  187. if s[0] != "/" and s[1] == ":":
  188. s = s[2:] # @@@ Hack when drive letter present
  189. return s
  190. CONTEXT = 0
  191. PRED = 1
  192. SUBJ = 2
  193. OBJ = 3
  194. PARTS = PRED, SUBJ, OBJ
  195. ALL4 = CONTEXT, PRED, SUBJ, OBJ
  196. SYMBOL = 0
  197. FORMULA = 1
  198. LITERAL = 2
  199. LITERAL_DT = 21
  200. LITERAL_LANG = 22
  201. ANONYMOUS = 3
  202. XMLLITERAL = 25
  203. Logic_NS = "http://www.w3.org/2000/10/swap/log#"
  204. NODE_MERGE_URI = Logic_NS + "is" # Pseudo-property indicating node merging
  205. forSomeSym = Logic_NS + "forSome"
  206. forAllSym = Logic_NS + "forAll"
  207. RDF_type_URI = "http://www.w3.org/1999/02/22-rdf-syntax-ns#type"
  208. RDF_NS_URI = "http://www.w3.org/1999/02/22-rdf-syntax-ns#"
  209. OWL_NS = "http://www.w3.org/2002/07/owl#"
  210. DAML_sameAs_URI = OWL_NS + "sameAs"
  211. parsesTo_URI = Logic_NS + "parsesTo"
  212. RDF_spec = "http://www.w3.org/TR/REC-rdf-syntax/"
  213. List_NS = RDF_NS_URI # From 20030808
  214. _Old_Logic_NS = "http://www.w3.org/2000/10/swap/log.n3#"
  215. N3_first = (SYMBOL, List_NS + "first")
  216. N3_rest = (SYMBOL, List_NS + "rest")
  217. N3_li = (SYMBOL, List_NS + "li")
  218. N3_nil = (SYMBOL, List_NS + "nil")
  219. N3_List = (SYMBOL, List_NS + "List")
  220. N3_Empty = (SYMBOL, List_NS + "Empty")
  221. runNamespaceValue: Optional[str] = None
  222. def runNamespace() -> str:
  223. """Returns a URI suitable as a namespace for run-local objects"""
  224. # @@@ include hostname (privacy?) (hash it?)
  225. global runNamespaceValue
  226. if runNamespaceValue is None:
  227. runNamespaceValue = join(base(), _unique_id()) + "#"
  228. return runNamespaceValue
  229. nextu = 0
  230. def uniqueURI() -> str:
  231. """A unique URI"""
  232. global nextu
  233. nextu += 1
  234. return runNamespace() + "u_" + str(nextu)
  235. tracking = False
  236. chatty_flag = 50
  237. # from why import BecauseOfData, becauseSubexpression
  238. def BecauseOfData(*args: Any, **kargs: Any) -> None:
  239. # print args, kargs
  240. pass
  241. def becauseSubexpression(*args: Any, **kargs: Any) -> None:
  242. # print args, kargs
  243. pass
  244. N3_forSome_URI = forSomeSym
  245. N3_forAll_URI = forAllSym
  246. # Magic resources we know about
  247. ADDED_HASH = "#" # Stop where we use this in case we want to remove it!
  248. # This is the hash on namespace URIs
  249. RDF_type = (SYMBOL, RDF_type_URI)
  250. DAML_sameAs = (SYMBOL, DAML_sameAs_URI)
  251. LOG_implies_URI = "http://www.w3.org/2000/10/swap/log#implies"
  252. BOOLEAN_DATATYPE = _XSD_PFX + "boolean"
  253. DECIMAL_DATATYPE = _XSD_PFX + "decimal"
  254. DOUBLE_DATATYPE = _XSD_PFX + "double"
  255. FLOAT_DATATYPE = _XSD_PFX + "float"
  256. INTEGER_DATATYPE = _XSD_PFX + "integer"
  257. option_noregen = 0 # If set, do not regenerate genids on output
  258. # @@ I18n - the notname chars need extending for well known unicode non-text
  259. # characters. The XML spec switched to assuming unknown things were name
  260. # characters.
  261. # _namechars = string.lowercase + string.uppercase + string.digits + '_-'
  262. _notQNameChars = set("\t\r\n !\"#$&'()*,+/;<=>?@[\\]^`{|}~") # else valid qname :-/
  263. _notKeywordsChars = _notQNameChars | {"."}
  264. _notNameChars = _notQNameChars | {":"} # Assume anything else valid name :-/
  265. _rdfns = "http://www.w3.org/1999/02/22-rdf-syntax-ns#"
  266. hexChars = set("ABCDEFabcdef0123456789")
  267. escapeChars = set("(_~.-!$&'()*+,;=/?#@%)") # valid for \ escapes in localnames
  268. numberChars = set("0123456789-")
  269. numberCharsPlus = numberChars | {"+", "."}
  270. def unicodeExpand(m: Match) -> str:
  271. try:
  272. return chr(int(m.group(1), 16))
  273. except Exception:
  274. raise Exception("Invalid unicode code point: " + m.group(1))
  275. unicodeEscape4 = re.compile(r"\\u([0-9a-fA-F]{4})")
  276. unicodeEscape8 = re.compile(r"\\U([0-9a-fA-F]{8})")
  277. N3CommentCharacter = "#" # For unix script # ! compatibility
  278. # Parse string to sink
  279. #
  280. # Regular expressions:
  281. eol = re.compile(r"[ \t]*(#[^\n]*)?\r?\n") # end of line, poss. w/comment
  282. eof = re.compile(r"[ \t]*(#[^\n]*)?$") # end of file, poss. w/comment
  283. ws = re.compile(r"[ \t]*") # Whitespace not including NL
  284. signed_integer = re.compile(r"[-+]?[0-9]+") # integer
  285. integer_syntax = re.compile(r"[-+]?[0-9]+")
  286. decimal_syntax = re.compile(r"[-+]?[0-9]*\.[0-9]+")
  287. exponent_syntax = re.compile(
  288. r"[-+]?(?:[0-9]+\.[0-9]*|\.[0-9]+|[0-9]+)(?:e|E)[-+]?[0-9]+"
  289. )
  290. digitstring = re.compile(r"[0-9]+") # Unsigned integer
  291. interesting = re.compile(r"""[\\\r\n\"\']""")
  292. langcode = re.compile(r"[a-zA-Z0-9]+(-[a-zA-Z0-9]+)*")
  293. class sfloat(str): # noqa: N801
  294. """don't normalize raw XSD.double string representation"""
  295. class SinkParser:
  296. def __init__(
  297. self,
  298. store: RDFSink,
  299. openFormula: Optional[Formula] = None,
  300. thisDoc: str = "",
  301. baseURI: Optional[str] = None,
  302. genPrefix: str = "",
  303. why: Optional[Callable[[], None]] = None,
  304. turtle: bool = False,
  305. ):
  306. """note: namespace names should *not* end in # ;
  307. the # will get added during qname processing"""
  308. self._bindings = {}
  309. if thisDoc != "":
  310. assert ":" in thisDoc, "Document URI not absolute: <%s>" % thisDoc
  311. self._bindings[""] = thisDoc + "#" # default
  312. self._store = store
  313. if genPrefix:
  314. # TODO FIXME: there is no function named setGenPrefix
  315. store.setGenPrefix(genPrefix) # type: ignore[attr-defined] # pass it on
  316. self._thisDoc = thisDoc
  317. self.lines = 0 # for error handling
  318. self.startOfLine = 0 # For calculating character number
  319. self._genPrefix = genPrefix
  320. self.keywords = ["a", "this", "bind", "has", "is", "of", "true", "false"]
  321. self.keywordsSet = 0 # Then only can others be considered qnames
  322. self._anonymousNodes: Dict[str, BNode] = {}
  323. # Dict of anon nodes already declared ln: Term
  324. self._variables: Dict[str, Variable] = {}
  325. self._parentVariables: Dict[str, Variable] = {}
  326. self._reason = why # Why the parser was asked to parse this
  327. self.turtle = turtle # raise exception when encountering N3 extensions
  328. # Turtle allows single or double quotes around strings, whereas N3
  329. # only allows double quotes.
  330. self.string_delimiters = ('"', "'") if turtle else ('"',)
  331. self._reason2: Optional[Callable[..., None]] = None # Why these triples
  332. # was: diag.tracking
  333. if tracking:
  334. # type error: "BecauseOfData" does not return a value
  335. self._reason2 = BecauseOfData( # type: ignore[func-returns-value]
  336. store.newSymbol(thisDoc), because=self._reason
  337. )
  338. self._baseURI: Optional[str]
  339. if baseURI:
  340. self._baseURI = baseURI
  341. else:
  342. if thisDoc:
  343. self._baseURI = thisDoc
  344. else:
  345. self._baseURI = None
  346. assert not self._baseURI or ":" in self._baseURI
  347. if not self._genPrefix:
  348. if self._thisDoc:
  349. self._genPrefix = self._thisDoc + "#_g"
  350. else:
  351. self._genPrefix = uniqueURI()
  352. self._formula: Optional[Formula]
  353. if openFormula is None and not turtle:
  354. if self._thisDoc:
  355. # TODO FIXME: store.newFormula does not take any arguments
  356. self._formula = store.newFormula(thisDoc + "#_formula") # type: ignore[call-arg]
  357. else:
  358. self._formula = store.newFormula()
  359. else:
  360. self._formula = openFormula
  361. self._context: Optional[Formula] = self._formula
  362. self._parentContext: Optional[Formula] = None
  363. def here(self, i: int) -> str:
  364. """String generated from position in file
  365. This is for repeatability when referring people to bnodes in a document.
  366. This has diagnostic uses less formally, as it should point one to which
  367. bnode the arbitrary identifier actually is. It gives the
  368. line and character number of the '[' charcacter or path character
  369. which introduced the blank node. The first blank node is boringly
  370. _L1C1. It used to be used only for tracking, but for tests in general
  371. it makes the canonical ordering of bnodes repeatable."""
  372. return "%s_L%iC%i" % (self._genPrefix, self.lines, i - self.startOfLine + 1)
  373. def formula(self) -> Optional[Formula]:
  374. return self._formula
  375. def loadStream(self, stream: Union[IO[str], IO[bytes]]) -> Optional[Formula]:
  376. return self.loadBuf(stream.read()) # Not ideal
  377. def loadBuf(self, buf: Union[str, bytes]) -> Optional[Formula]:
  378. """Parses a buffer and returns its top level formula"""
  379. self.startDoc()
  380. self.feed(buf)
  381. return self.endDoc() # self._formula
  382. def feed(self, octets: Union[str, bytes]) -> None:
  383. """Feed an octet stream to the parser
  384. if BadSyntax is raised, the string
  385. passed in the exception object is the
  386. remainder after any statements have been parsed.
  387. So if there is more data to feed to the
  388. parser, it should be straightforward to recover."""
  389. if not isinstance(octets, str):
  390. s = octets.decode("utf-8")
  391. # NB already decoded, so \ufeff
  392. if len(s) > 0 and s[0] == codecs.BOM_UTF8.decode("utf-8"):
  393. s = s[1:]
  394. else:
  395. s = octets
  396. i = 0
  397. while i >= 0:
  398. j = self.skipSpace(s, i)
  399. if j < 0:
  400. return
  401. i = self.directiveOrStatement(s, j)
  402. if i < 0:
  403. # print("# next char: %s" % s[j])
  404. self.BadSyntax(s, j, "expected directive or statement")
  405. def directiveOrStatement(self, argstr: str, h: int) -> int:
  406. i = self.skipSpace(argstr, h)
  407. if i < 0:
  408. return i # EOF
  409. if self.turtle:
  410. j = self.sparqlDirective(argstr, i)
  411. if j >= 0:
  412. return j
  413. j = self.directive(argstr, i)
  414. if j >= 0:
  415. return self.checkDot(argstr, j)
  416. j = self.statement(argstr, i)
  417. if j >= 0:
  418. return self.checkDot(argstr, j)
  419. return j
  420. # @@I18N
  421. # _namechars = string.lowercase + string.uppercase + string.digits + '_-'
  422. def tok(self, tok: str, argstr: str, i: int, colon: bool = False) -> int:
  423. """Check for keyword. Space must have been stripped on entry and
  424. we must not be at end of file.
  425. if colon, then keyword followed by colon is ok
  426. (`@prefix:<blah>` is ok, rdf:type shortcut a must be followed by ws)
  427. """
  428. assert tok[0] not in _notNameChars # not for punctuation
  429. if argstr[i] == "@":
  430. i += 1
  431. else:
  432. if tok not in self.keywords:
  433. return -1 # No, this has neither keywords declaration nor "@"
  434. i_plus_len_tok = i + len(tok)
  435. if (
  436. argstr[i:i_plus_len_tok] == tok
  437. and (argstr[i_plus_len_tok] in _notKeywordsChars)
  438. or (colon and argstr[i_plus_len_tok] == ":")
  439. ):
  440. return i_plus_len_tok
  441. else:
  442. return -1
  443. def sparqlTok(self, tok: str, argstr: str, i: int) -> int:
  444. """Check for SPARQL keyword. Space must have been stripped on entry
  445. and we must not be at end of file.
  446. Case insensitive and not preceded by @
  447. """
  448. assert tok[0] not in _notNameChars # not for punctuation
  449. len_tok = len(tok)
  450. if argstr[i : i + len_tok].lower() == tok.lower() and (
  451. argstr[i + len_tok] in _notQNameChars
  452. ):
  453. i += len_tok
  454. return i
  455. else:
  456. return -1
  457. def directive(self, argstr: str, i: int) -> int:
  458. j = self.skipSpace(argstr, i)
  459. if j < 0:
  460. return j # eof
  461. res: typing.List[str] = []
  462. j = self.tok("bind", argstr, i) # implied "#". Obsolete.
  463. if j > 0:
  464. self.BadSyntax(argstr, i, "keyword bind is obsolete: use @prefix")
  465. j = self.tok("keywords", argstr, i)
  466. if j > 0:
  467. if self.turtle:
  468. self.BadSyntax(argstr, i, "Found 'keywords' when in Turtle mode.")
  469. i = self.commaSeparatedList(argstr, j, res, self.bareWord)
  470. if i < 0:
  471. self.BadSyntax(
  472. argstr, i, "'@keywords' needs comma separated list of words"
  473. )
  474. self.setKeywords(res[:])
  475. return i
  476. j = self.tok("forAll", argstr, i)
  477. if j > 0:
  478. if self.turtle:
  479. self.BadSyntax(argstr, i, "Found 'forAll' when in Turtle mode.")
  480. i = self.commaSeparatedList(argstr, j, res, self.uri_ref2)
  481. if i < 0:
  482. self.BadSyntax(argstr, i, "Bad variable list after @forAll")
  483. for x in res:
  484. # self._context.declareUniversal(x)
  485. if x not in self._variables or x in self._parentVariables:
  486. # type error: Item "None" of "Optional[Formula]" has no attribute "newUniversal"
  487. self._variables[x] = self._context.newUniversal(x) # type: ignore[union-attr]
  488. return i
  489. j = self.tok("forSome", argstr, i)
  490. if j > 0:
  491. if self.turtle:
  492. self.BadSyntax(argstr, i, "Found 'forSome' when in Turtle mode.")
  493. i = self.commaSeparatedList(argstr, j, res, self.uri_ref2)
  494. if i < 0:
  495. self.BadSyntax(argstr, i, "Bad variable list after @forSome")
  496. for x in res:
  497. # type error: Item "None" of "Optional[Formula]" has no attribute "declareExistential"
  498. self._context.declareExistential(x) # type: ignore[union-attr]
  499. return i
  500. j = self.tok("prefix", argstr, i, colon=True) # no implied "#"
  501. if j >= 0:
  502. t: typing.List[Union[Identifier, Tuple[str, str]]] = []
  503. i = self.qname(argstr, j, t)
  504. if i < 0:
  505. self.BadSyntax(argstr, j, "expected qname after @prefix")
  506. j = self.uri_ref2(argstr, i, t)
  507. if j < 0:
  508. self.BadSyntax(argstr, i, "expected <uriref> after @prefix _qname_")
  509. ns: str = self.uriOf(t[1])
  510. if self._baseURI:
  511. ns = join(self._baseURI, ns)
  512. elif ":" not in ns:
  513. self.BadSyntax(
  514. argstr,
  515. j,
  516. f"With no base URI, cannot use relative URI in @prefix <{ns}>",
  517. )
  518. assert ":" in ns # must be absolute
  519. self._bindings[t[0][0]] = ns
  520. self.bind(t[0][0], hexify(ns))
  521. return j
  522. j = self.tok("base", argstr, i) # Added 2007/7/7
  523. if j >= 0:
  524. t = []
  525. i = self.uri_ref2(argstr, j, t)
  526. if i < 0:
  527. self.BadSyntax(argstr, j, "expected <uri> after @base ")
  528. ns = self.uriOf(t[0])
  529. if self._baseURI:
  530. ns = join(self._baseURI, ns)
  531. else:
  532. self.BadSyntax(
  533. argstr,
  534. j,
  535. "With no previous base URI, cannot use "
  536. + "relative URI in @base <"
  537. + ns
  538. + ">",
  539. )
  540. assert ":" in ns # must be absolute
  541. self._baseURI = ns
  542. return i
  543. return -1 # Not a directive, could be something else.
  544. def sparqlDirective(self, argstr: str, i: int) -> int:
  545. """
  546. turtle and trig support BASE/PREFIX without @ and without
  547. terminating .
  548. """
  549. j = self.skipSpace(argstr, i)
  550. if j < 0:
  551. return j # eof
  552. j = self.sparqlTok("PREFIX", argstr, i)
  553. if j >= 0:
  554. t: typing.List[Any] = []
  555. i = self.qname(argstr, j, t)
  556. if i < 0:
  557. self.BadSyntax(argstr, j, "expected qname after @prefix")
  558. j = self.uri_ref2(argstr, i, t)
  559. if j < 0:
  560. self.BadSyntax(argstr, i, "expected <uriref> after @prefix _qname_")
  561. ns = self.uriOf(t[1])
  562. if self._baseURI:
  563. ns = join(self._baseURI, ns)
  564. elif ":" not in ns:
  565. self.BadSyntax(
  566. argstr,
  567. j,
  568. "With no base URI, cannot use "
  569. + "relative URI in @prefix <"
  570. + ns
  571. + ">",
  572. )
  573. assert ":" in ns # must be absolute
  574. self._bindings[t[0][0]] = ns
  575. self.bind(t[0][0], hexify(ns))
  576. return j
  577. j = self.sparqlTok("BASE", argstr, i)
  578. if j >= 0:
  579. t = []
  580. i = self.uri_ref2(argstr, j, t)
  581. if i < 0:
  582. self.BadSyntax(argstr, j, "expected <uri> after @base ")
  583. ns = self.uriOf(t[0])
  584. if self._baseURI:
  585. ns = join(self._baseURI, ns)
  586. else:
  587. self.BadSyntax(
  588. argstr,
  589. j,
  590. "With no previous base URI, cannot use "
  591. + "relative URI in @base <"
  592. + ns
  593. + ">",
  594. )
  595. assert ":" in ns # must be absolute
  596. self._baseURI = ns
  597. return i
  598. return -1 # Not a directive, could be something else.
  599. def bind(self, qn: str, uri: bytes) -> None:
  600. assert isinstance(uri, bytes), "Any unicode must be %x-encoded already"
  601. if qn == "":
  602. self._store.setDefaultNamespace(uri)
  603. else:
  604. self._store.bind(qn, uri)
  605. def setKeywords(self, k: Optional[typing.List[str]]) -> None:
  606. """Takes a list of strings"""
  607. if k is None:
  608. self.keywordsSet = 0
  609. else:
  610. self.keywords = k
  611. self.keywordsSet = 1
  612. def startDoc(self) -> None:
  613. # was: self._store.startDoc()
  614. self._store.startDoc(self._formula)
  615. def endDoc(self) -> Optional[Formula]:
  616. """Signal end of document and stop parsing. returns formula"""
  617. self._store.endDoc(self._formula) # don't canonicalize yet
  618. return self._formula
  619. def makeStatement(self, quadruple) -> None:
  620. # $$$$$$$$$$$$$$$$$$$$$
  621. # print "# Parser output: ", `quadruple`
  622. self._store.makeStatement(quadruple, why=self._reason2)
  623. def statement(self, argstr: str, i: int) -> int:
  624. r: typing.List[Any] = []
  625. i = self.object(argstr, i, r) # Allow literal for subject - extends RDF
  626. if i < 0:
  627. return i
  628. j = self.property_list(argstr, i, r[0])
  629. if j < 0:
  630. self.BadSyntax(argstr, i, "expected propertylist")
  631. return j
  632. def subject(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  633. return self.item(argstr, i, res)
  634. def verb(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  635. """has _prop_
  636. is _prop_ of
  637. a
  638. =
  639. _prop_
  640. >- prop ->
  641. <- prop -<
  642. _operator_"""
  643. j = self.skipSpace(argstr, i)
  644. if j < 0:
  645. return j # eof
  646. r: typing.List[Any] = []
  647. j = self.tok("has", argstr, i)
  648. if j >= 0:
  649. if self.turtle:
  650. self.BadSyntax(argstr, i, "Found 'has' keyword in Turtle mode")
  651. i = self.prop(argstr, j, r)
  652. if i < 0:
  653. self.BadSyntax(argstr, j, "expected property after 'has'")
  654. res.append(("->", r[0]))
  655. return i
  656. j = self.tok("is", argstr, i)
  657. if j >= 0:
  658. if self.turtle:
  659. self.BadSyntax(argstr, i, "Found 'is' keyword in Turtle mode")
  660. i = self.prop(argstr, j, r)
  661. if i < 0:
  662. self.BadSyntax(argstr, j, "expected <property> after 'is'")
  663. j = self.skipSpace(argstr, i)
  664. if j < 0:
  665. self.BadSyntax(
  666. argstr, i, "End of file found, expected property after 'is'"
  667. )
  668. i = j
  669. j = self.tok("of", argstr, i)
  670. if j < 0:
  671. self.BadSyntax(argstr, i, "expected 'of' after 'is' <prop>")
  672. res.append(("<-", r[0]))
  673. return j
  674. j = self.tok("a", argstr, i)
  675. if j >= 0:
  676. res.append(("->", RDF_type))
  677. return j
  678. if argstr[i : i + 2] == "<=":
  679. if self.turtle:
  680. self.BadSyntax(argstr, i, "Found '<=' in Turtle mode. ")
  681. res.append(("<-", self._store.newSymbol(Logic_NS + "implies")))
  682. return i + 2
  683. if argstr[i] == "=":
  684. if self.turtle:
  685. self.BadSyntax(argstr, i, "Found '=' in Turtle mode")
  686. if argstr[i + 1] == ">":
  687. res.append(("->", self._store.newSymbol(Logic_NS + "implies")))
  688. return i + 2
  689. res.append(("->", DAML_sameAs))
  690. return i + 1
  691. if argstr[i : i + 2] == ":=":
  692. if self.turtle:
  693. self.BadSyntax(argstr, i, "Found ':=' in Turtle mode")
  694. # patch file relates two formulae, uses this @@ really?
  695. res.append(("->", Logic_NS + "becomes"))
  696. return i + 2
  697. j = self.prop(argstr, i, r)
  698. if j >= 0:
  699. res.append(("->", r[0]))
  700. return j
  701. if argstr[i : i + 2] == ">-" or argstr[i : i + 2] == "<-":
  702. self.BadSyntax(argstr, j, ">- ... -> syntax is obsolete.")
  703. return -1
  704. def prop(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  705. return self.item(argstr, i, res)
  706. def item(self, argstr: str, i, res: MutableSequence[Any]) -> int:
  707. return self.path(argstr, i, res)
  708. def blankNode(self, uri: Optional[str] = None) -> BNode:
  709. return self._store.newBlankNode(self._context, uri, why=self._reason2)
  710. def path(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  711. """Parse the path production."""
  712. j = self.nodeOrLiteral(argstr, i, res)
  713. if j < 0:
  714. return j # nope
  715. while argstr[j] in {"!", "^"}: # no spaces, must follow exactly (?)
  716. ch = argstr[j]
  717. subj = res.pop()
  718. obj = self.blankNode(uri=self.here(j))
  719. j = self.node(argstr, j + 1, res)
  720. if j < 0:
  721. self.BadSyntax(argstr, j, "EOF found in middle of path syntax")
  722. pred = res.pop()
  723. if ch == "^": # Reverse traverse
  724. self.makeStatement((self._context, pred, obj, subj))
  725. else:
  726. self.makeStatement((self._context, pred, subj, obj))
  727. res.append(obj)
  728. return j
  729. def anonymousNode(self, ln: str) -> BNode:
  730. """Remember or generate a term for one of these _: anonymous nodes"""
  731. term = self._anonymousNodes.get(ln, None)
  732. if term is not None:
  733. return term
  734. term = self._store.newBlankNode(self._context, why=self._reason2)
  735. self._anonymousNodes[ln] = term
  736. return term
  737. def node(
  738. self,
  739. argstr: str,
  740. i: int,
  741. res: MutableSequence[Any],
  742. subjectAlready: Optional[Node] = None,
  743. ) -> int:
  744. """Parse the <node> production.
  745. Space is now skipped once at the beginning
  746. instead of in multiple calls to self.skipSpace().
  747. """
  748. subj: Optional[Node] = subjectAlready
  749. j = self.skipSpace(argstr, i)
  750. if j < 0:
  751. return j # eof
  752. i = j
  753. ch = argstr[i] # Quick 1-character checks first:
  754. if ch == "[":
  755. bnodeID = self.here(i)
  756. j = self.skipSpace(argstr, i + 1)
  757. if j < 0:
  758. self.BadSyntax(argstr, i, "EOF after '['")
  759. # Hack for "is" binding name to anon node
  760. if argstr[j] == "=":
  761. if self.turtle:
  762. self.BadSyntax(
  763. argstr, j, "Found '[=' or '[ =' when in turtle mode."
  764. )
  765. i = j + 1
  766. objs: typing.List[Node] = []
  767. j = self.objectList(argstr, i, objs)
  768. if j >= 0:
  769. subj = objs[0]
  770. if len(objs) > 1:
  771. for obj in objs:
  772. self.makeStatement((self._context, DAML_sameAs, subj, obj))
  773. j = self.skipSpace(argstr, j)
  774. if j < 0:
  775. self.BadSyntax(
  776. argstr, i, "EOF when objectList expected after [ = "
  777. )
  778. if argstr[j] == ";":
  779. j += 1
  780. else:
  781. self.BadSyntax(argstr, i, "objectList expected after [= ")
  782. if subj is None:
  783. subj = self.blankNode(uri=bnodeID)
  784. i = self.property_list(argstr, j, subj)
  785. if i < 0:
  786. self.BadSyntax(argstr, j, "property_list expected")
  787. j = self.skipSpace(argstr, i)
  788. if j < 0:
  789. self.BadSyntax(
  790. argstr, i, "EOF when ']' expected after [ <propertyList>"
  791. )
  792. if argstr[j] != "]":
  793. self.BadSyntax(argstr, j, "']' expected")
  794. res.append(subj)
  795. return j + 1
  796. if not self.turtle and ch == "{":
  797. # if self.turtle:
  798. # self.BadSyntax(argstr, i,
  799. # "found '{' while in Turtle mode, Formulas not supported!")
  800. ch2 = argstr[i + 1]
  801. if ch2 == "$":
  802. # a set
  803. i += 1
  804. j = i + 1
  805. List = []
  806. first_run = True
  807. while 1:
  808. i = self.skipSpace(argstr, j)
  809. if i < 0:
  810. self.BadSyntax(argstr, i, "needed '$}', found end.")
  811. if argstr[i : i + 2] == "$}":
  812. j = i + 2
  813. break
  814. if not first_run:
  815. if argstr[i] == ",":
  816. i += 1
  817. else:
  818. self.BadSyntax(argstr, i, "expected: ','")
  819. else:
  820. first_run = False
  821. item: typing.List[Any] = []
  822. j = self.item(argstr, i, item) # @@@@@ should be path, was object
  823. if j < 0:
  824. self.BadSyntax(argstr, i, "expected item in set or '$}'")
  825. List.append(self._store.intern(item[0]))
  826. res.append(self._store.newSet(List, self._context))
  827. return j
  828. else:
  829. # parse a formula
  830. j = i + 1
  831. oldParentContext = self._parentContext
  832. self._parentContext = self._context
  833. parentAnonymousNodes = self._anonymousNodes
  834. grandParentVariables = self._parentVariables
  835. self._parentVariables = self._variables
  836. self._anonymousNodes = {}
  837. self._variables = self._variables.copy()
  838. reason2 = self._reason2
  839. self._reason2 = becauseSubexpression
  840. if subj is None:
  841. # type error: Incompatible types in assignment (expression has type "Formula", variable has type "Optional[Node]")
  842. subj = self._store.newFormula() # type: ignore[assignment]
  843. # type error: Incompatible types in assignment (expression has type "Optional[Node]", variable has type "Optional[Formula]")
  844. self._context = subj # type: ignore[assignment]
  845. while 1:
  846. i = self.skipSpace(argstr, j)
  847. if i < 0:
  848. self.BadSyntax(argstr, i, "needed '}', found end.")
  849. if argstr[i] == "}":
  850. j = i + 1
  851. break
  852. j = self.directiveOrStatement(argstr, i)
  853. if j < 0:
  854. self.BadSyntax(argstr, i, "expected statement or '}'")
  855. self._anonymousNodes = parentAnonymousNodes
  856. self._variables = self._parentVariables
  857. self._parentVariables = grandParentVariables
  858. self._context = self._parentContext
  859. self._reason2 = reason2
  860. self._parentContext = oldParentContext
  861. # type error: Item "Node" of "Optional[Node]" has no attribute "close"
  862. res.append(
  863. subj.close() # type: ignore[union-attr]
  864. ) # No use until closed
  865. return j
  866. if ch == "(":
  867. thing_type: Callable[
  868. [typing.List[Any], Optional[Formula]], Union[Set[Any], IdentifiedNode]
  869. ]
  870. thing_type = self._store.newList
  871. ch2 = argstr[i + 1]
  872. if ch2 == "$":
  873. thing_type = self._store.newSet
  874. i += 1
  875. j = i + 1
  876. List = []
  877. while 1:
  878. i = self.skipSpace(argstr, j)
  879. if i < 0:
  880. self.BadSyntax(argstr, i, "needed ')', found end.")
  881. if argstr[i] == ")":
  882. j = i + 1
  883. break
  884. item = []
  885. j = self.item(argstr, i, item) # @@@@@ should be path, was object
  886. if j < 0:
  887. self.BadSyntax(argstr, i, "expected item in list or ')'")
  888. List.append(self._store.intern(item[0]))
  889. res.append(thing_type(List, self._context))
  890. return j
  891. j = self.tok("this", argstr, i) # This context
  892. if j >= 0:
  893. self.BadSyntax(
  894. argstr,
  895. i,
  896. "Keyword 'this' was ancient N3. Now use "
  897. + "@forSome and @forAll keywords.",
  898. )
  899. # booleans
  900. j = self.tok("true", argstr, i)
  901. if j >= 0:
  902. res.append(True)
  903. return j
  904. j = self.tok("false", argstr, i)
  905. if j >= 0:
  906. res.append(False)
  907. return j
  908. if subj is None: # If this can be a named node, then check for a name.
  909. j = self.uri_ref2(argstr, i, res)
  910. if j >= 0:
  911. return j
  912. return -1
  913. def property_list(self, argstr: str, i: int, subj: Node) -> int:
  914. """Parse property list
  915. Leaves the terminating punctuation in the buffer
  916. """
  917. while 1:
  918. while 1: # skip repeat ;
  919. j = self.skipSpace(argstr, i)
  920. if j < 0:
  921. self.BadSyntax(
  922. argstr, i, "EOF found when expected verb in property list"
  923. )
  924. if argstr[j] != ";":
  925. break
  926. i = j + 1
  927. if argstr[j : j + 2] == ":-":
  928. if self.turtle:
  929. self.BadSyntax(argstr, j, "Found in ':-' in Turtle mode")
  930. i = j + 2
  931. res: typing.List[Any] = []
  932. j = self.node(argstr, i, res, subj)
  933. if j < 0:
  934. self.BadSyntax(argstr, i, "bad {} or () or [] node after :- ")
  935. i = j
  936. continue
  937. i = j
  938. v: typing.List[Any] = []
  939. j = self.verb(argstr, i, v)
  940. if j <= 0:
  941. return i # void but valid
  942. objs: typing.List[Any] = []
  943. i = self.objectList(argstr, j, objs)
  944. if i < 0:
  945. self.BadSyntax(argstr, j, "objectList expected")
  946. for obj in objs:
  947. dira, sym = v[0]
  948. if dira == "->":
  949. self.makeStatement((self._context, sym, subj, obj))
  950. else:
  951. self.makeStatement((self._context, sym, obj, subj))
  952. j = self.skipSpace(argstr, i)
  953. if j < 0:
  954. self.BadSyntax(argstr, j, "EOF found in list of objects")
  955. if argstr[i] != ";":
  956. return i
  957. i += 1 # skip semicolon and continue
  958. def commaSeparatedList(
  959. self,
  960. argstr: str,
  961. j: int,
  962. res: MutableSequence[Any],
  963. what: Callable[[str, int, MutableSequence[Any]], int],
  964. ) -> int:
  965. """return value: -1 bad syntax; >1 new position in argstr
  966. res has things found appended
  967. """
  968. i = self.skipSpace(argstr, j)
  969. if i < 0:
  970. self.BadSyntax(argstr, i, "EOF found expecting comma sep list")
  971. if argstr[i] == ".":
  972. return j # empty list is OK
  973. i = what(argstr, i, res)
  974. if i < 0:
  975. return -1
  976. while 1:
  977. j = self.skipSpace(argstr, i)
  978. if j < 0:
  979. return j # eof
  980. ch = argstr[j]
  981. if ch != ",":
  982. if ch != ".":
  983. return -1
  984. return j # Found but not swallowed "."
  985. i = what(argstr, j + 1, res)
  986. if i < 0:
  987. self.BadSyntax(argstr, i, "bad list content")
  988. def objectList(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  989. i = self.object(argstr, i, res)
  990. if i < 0:
  991. return -1
  992. while 1:
  993. j = self.skipSpace(argstr, i)
  994. if j < 0:
  995. self.BadSyntax(argstr, j, "EOF found after object")
  996. if argstr[j] != ",":
  997. return j # Found something else!
  998. i = self.object(argstr, j + 1, res)
  999. if i < 0:
  1000. return i
  1001. def checkDot(self, argstr: str, i: int) -> int:
  1002. j = self.skipSpace(argstr, i)
  1003. if j < 0:
  1004. return j # eof
  1005. ch = argstr[j]
  1006. if ch == ".":
  1007. return j + 1 # skip
  1008. if ch == "}":
  1009. return j # don't skip it
  1010. if ch == "]":
  1011. return j
  1012. self.BadSyntax(argstr, j, "expected '.' or '}' or ']' at end of statement")
  1013. def uri_ref2(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  1014. """Generate uri from n3 representation.
  1015. Note that the RDF convention of directly concatenating
  1016. NS and local name is now used though I prefer inserting a '#'
  1017. to make the namesapces look more like what XML folks expect.
  1018. """
  1019. qn: typing.List[Any] = []
  1020. j = self.qname(argstr, i, qn)
  1021. if j >= 0:
  1022. pfx, ln = qn[0]
  1023. if pfx is None:
  1024. assert 0, "not used?"
  1025. ns = self._baseURI + ADDED_HASH # type: ignore[unreachable]
  1026. else:
  1027. try:
  1028. ns = self._bindings[pfx]
  1029. except KeyError:
  1030. if pfx == "_": # Magic prefix 2001/05/30, can be changed
  1031. res.append(self.anonymousNode(ln))
  1032. return j
  1033. if not self.turtle and pfx == "":
  1034. ns = join(self._baseURI or "", "#")
  1035. else:
  1036. self.BadSyntax(argstr, i, 'Prefix "%s:" not bound' % (pfx))
  1037. symb = self._store.newSymbol(ns + ln)
  1038. res.append(self._variables.get(symb, symb))
  1039. return j
  1040. i = self.skipSpace(argstr, i)
  1041. if i < 0:
  1042. return -1
  1043. if argstr[i] == "?":
  1044. v: typing.List[Any] = []
  1045. j = self.variable(argstr, i, v)
  1046. if j > 0: # Forget variables as a class, only in context.
  1047. res.append(v[0])
  1048. return j
  1049. return -1
  1050. elif argstr[i] == "<":
  1051. st = i + 1
  1052. i = argstr.find(">", st)
  1053. if i >= 0:
  1054. uref = argstr[st:i] # the join should dealt with "":
  1055. # expand unicode escapes
  1056. uref = unicodeEscape8.sub(unicodeExpand, uref)
  1057. uref = unicodeEscape4.sub(unicodeExpand, uref)
  1058. if self._baseURI:
  1059. uref = join(self._baseURI, uref) # was: uripath.join
  1060. else:
  1061. assert (
  1062. ":" in uref
  1063. ), "With no base URI, cannot deal with relative URIs"
  1064. if argstr[i - 1] == "#" and not uref[-1:] == "#":
  1065. uref += "#" # She meant it! Weirdness in urlparse?
  1066. symb = self._store.newSymbol(uref)
  1067. res.append(self._variables.get(symb, symb))
  1068. return i + 1
  1069. self.BadSyntax(argstr, j, "unterminated URI reference")
  1070. elif self.keywordsSet:
  1071. v = []
  1072. j = self.bareWord(argstr, i, v)
  1073. if j < 0:
  1074. return -1 # Forget variables as a class, only in context.
  1075. if v[0] in self.keywords:
  1076. self.BadSyntax(argstr, i, 'Keyword "%s" not allowed here.' % v[0])
  1077. res.append(self._store.newSymbol(self._bindings[""] + v[0]))
  1078. return j
  1079. else:
  1080. return -1
  1081. def skipSpace(self, argstr: str, i: int) -> int:
  1082. """Skip white space, newlines and comments.
  1083. return -1 if EOF, else position of first non-ws character"""
  1084. # Most common case is a non-commented line starting with few spaces and tabs.
  1085. try:
  1086. while True:
  1087. ch = argstr[i]
  1088. if ch in {" ", "\t"}:
  1089. i += 1
  1090. continue
  1091. elif ch not in {"#", "\r", "\n"}:
  1092. return i
  1093. break
  1094. except IndexError:
  1095. return -1
  1096. while 1:
  1097. m = eol.match(argstr, i)
  1098. if m is None:
  1099. break
  1100. self.lines += 1
  1101. self.startOfLine = i = m.end() # Point to first character unmatched
  1102. m = ws.match(argstr, i)
  1103. if m is not None:
  1104. i = m.end()
  1105. m = eof.match(argstr, i)
  1106. return i if m is None else -1
  1107. def variable(self, argstr: str, i: int, res) -> int:
  1108. """?abc -> variable(:abc)"""
  1109. j = self.skipSpace(argstr, i)
  1110. if j < 0:
  1111. return -1
  1112. if argstr[j] != "?":
  1113. return -1
  1114. j += 1
  1115. i = j
  1116. if argstr[j] in numberChars:
  1117. self.BadSyntax(argstr, j, "Variable name can't start with '%s'" % argstr[j])
  1118. len_argstr = len(argstr)
  1119. while i < len_argstr and argstr[i] not in _notKeywordsChars:
  1120. i += 1
  1121. if self._parentContext is None:
  1122. varURI = self._store.newSymbol(self._baseURI + "#" + argstr[j:i]) # type: ignore[operator]
  1123. if varURI not in self._variables:
  1124. # type error: Item "None" of "Optional[Formula]" has no attribute "newUniversal"
  1125. self._variables[varURI] = self._context.newUniversal( # type: ignore[union-attr]
  1126. varURI, why=self._reason2
  1127. )
  1128. res.append(self._variables[varURI])
  1129. return i
  1130. # @@ was:
  1131. # self.BadSyntax(argstr, j,
  1132. # "Can't use ?xxx syntax for variable in outermost level: %s"
  1133. # % argstr[j-1:i])
  1134. varURI = self._store.newSymbol(self._baseURI + "#" + argstr[j:i]) # type: ignore[operator]
  1135. if varURI not in self._parentVariables:
  1136. self._parentVariables[varURI] = self._parentContext.newUniversal(
  1137. varURI, why=self._reason2
  1138. )
  1139. res.append(self._parentVariables[varURI])
  1140. return i
  1141. def bareWord(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  1142. """abc -> :abc"""
  1143. j = self.skipSpace(argstr, i)
  1144. if j < 0:
  1145. return -1
  1146. if argstr[j] in numberChars or argstr[j] in _notKeywordsChars:
  1147. return -1
  1148. i = j
  1149. len_argstr = len(argstr)
  1150. while i < len_argstr and argstr[i] not in _notKeywordsChars:
  1151. i += 1
  1152. res.append(argstr[j:i])
  1153. return i
  1154. def qname(
  1155. self,
  1156. argstr: str,
  1157. i: int,
  1158. res: MutableSequence[Union[Identifier, Tuple[str, str]]],
  1159. ) -> int:
  1160. """
  1161. xyz:def -> ('xyz', 'def')
  1162. If not in keywords and keywordsSet: def -> ('', 'def')
  1163. :def -> ('', 'def')
  1164. """
  1165. i = self.skipSpace(argstr, i)
  1166. if i < 0:
  1167. return -1
  1168. c = argstr[i]
  1169. if c in numberCharsPlus:
  1170. return -1
  1171. len_argstr = len(argstr)
  1172. if c not in _notNameChars:
  1173. j = i
  1174. i += 1
  1175. try:
  1176. while argstr[i] not in _notNameChars:
  1177. i += 1
  1178. except IndexError:
  1179. pass # Very rare.
  1180. if argstr[i - 1] == ".": # qname cannot end with "."
  1181. i -= 1
  1182. if i == j:
  1183. return -1
  1184. ln = argstr[j:i]
  1185. else: # First character is non-alpha
  1186. ln = "" # Was: None - TBL (why? useful?)
  1187. if i < len_argstr and argstr[i] == ":":
  1188. pfx = ln
  1189. # bnodes names have different rules
  1190. if pfx == "_":
  1191. allowedChars = _notNameChars
  1192. else:
  1193. allowedChars = _notQNameChars
  1194. i += 1
  1195. lastslash = False
  1196. start = i
  1197. ln = ""
  1198. while i < len_argstr:
  1199. c = argstr[i]
  1200. if c == "\\" and not lastslash: # Very rare.
  1201. lastslash = True
  1202. if start < i:
  1203. ln += argstr[start:i]
  1204. start = i + 1
  1205. elif c not in allowedChars or lastslash: # Most common case is "a-zA-Z"
  1206. if lastslash:
  1207. if c not in escapeChars:
  1208. raise BadSyntax(
  1209. self._thisDoc,
  1210. self.lines,
  1211. argstr,
  1212. i,
  1213. "illegal escape " + c,
  1214. )
  1215. elif c == "%": # Very rare.
  1216. if (
  1217. argstr[i + 1] not in hexChars
  1218. or argstr[i + 2] not in hexChars
  1219. ):
  1220. raise BadSyntax(
  1221. self._thisDoc,
  1222. self.lines,
  1223. argstr,
  1224. i,
  1225. "illegal hex escape " + c,
  1226. )
  1227. lastslash = False
  1228. else:
  1229. break
  1230. i += 1
  1231. if lastslash:
  1232. raise BadSyntax(
  1233. self._thisDoc, self.lines, argstr, i, "qname cannot end with \\"
  1234. )
  1235. if argstr[i - 1] == ".":
  1236. # localname cannot end in .
  1237. if len(ln) == 0 and start == i:
  1238. return -1
  1239. i -= 1
  1240. if start < i:
  1241. ln += argstr[start:i]
  1242. res.append((pfx, ln))
  1243. return i
  1244. else: # delimiter was not ":"
  1245. if ln and self.keywordsSet and ln not in self.keywords:
  1246. res.append(("", ln))
  1247. return i
  1248. return -1
  1249. def object(
  1250. self,
  1251. argstr: str,
  1252. i: int,
  1253. res: MutableSequence[Any],
  1254. ) -> int:
  1255. j = self.subject(argstr, i, res)
  1256. if j >= 0:
  1257. return j
  1258. else:
  1259. j = self.skipSpace(argstr, i)
  1260. if j < 0:
  1261. return -1
  1262. else:
  1263. i = j
  1264. ch = argstr[i]
  1265. if ch in self.string_delimiters:
  1266. ch_three = ch * 3
  1267. if argstr[i : i + 3] == ch_three:
  1268. delim = ch_three
  1269. i += 3
  1270. else:
  1271. delim = ch
  1272. i += 1
  1273. j, s = self.strconst(argstr, i, delim)
  1274. res.append(self._store.newLiteral(s)) # type: ignore[call-arg] # TODO FIXME
  1275. return j
  1276. else:
  1277. return -1
  1278. def nodeOrLiteral(self, argstr: str, i: int, res: MutableSequence[Any]) -> int:
  1279. j = self.node(argstr, i, res)
  1280. startline = self.lines # Remember where for error messages
  1281. if j >= 0:
  1282. return j
  1283. else:
  1284. j = self.skipSpace(argstr, i)
  1285. if j < 0:
  1286. return -1
  1287. else:
  1288. i = j
  1289. ch = argstr[i]
  1290. if ch in numberCharsPlus:
  1291. m = exponent_syntax.match(argstr, i)
  1292. if m:
  1293. j = m.end()
  1294. res.append(sfloat(argstr[i:j]))
  1295. return j
  1296. m = decimal_syntax.match(argstr, i)
  1297. if m:
  1298. j = m.end()
  1299. res.append(Decimal(argstr[i:j]))
  1300. return j
  1301. m = integer_syntax.match(argstr, i)
  1302. if m:
  1303. j = m.end()
  1304. res.append(long_type(argstr[i:j]))
  1305. return j
  1306. # return -1 ## or fall through?
  1307. ch_three = ch * 3
  1308. if ch in self.string_delimiters:
  1309. if argstr[i : i + 3] == ch_three:
  1310. delim = ch_three
  1311. i += 3
  1312. else:
  1313. delim = ch
  1314. i += 1
  1315. dt = None
  1316. j, s = self.strconst(argstr, i, delim)
  1317. lang = None
  1318. if argstr[j] == "@": # Language?
  1319. m = langcode.match(argstr, j + 1)
  1320. if m is None:
  1321. raise BadSyntax(
  1322. self._thisDoc,
  1323. startline,
  1324. argstr,
  1325. i,
  1326. "Bad language code syntax on string " + "literal, after @",
  1327. )
  1328. i = m.end()
  1329. lang = argstr[j + 1 : i]
  1330. j = i
  1331. if argstr[j : j + 2] == "^^":
  1332. res2: typing.List[Any] = []
  1333. j = self.uri_ref2(argstr, j + 2, res2) # Read datatype URI
  1334. dt = res2[0]
  1335. res.append(self._store.newLiteral(s, dt, lang))
  1336. return j
  1337. else:
  1338. return -1
  1339. def uriOf(self, sym: Union[Identifier, Tuple[str, str]]) -> str:
  1340. if isinstance(sym, tuple):
  1341. return sym[1] # old system for --pipe
  1342. # return sym.uriref() # cwm api
  1343. return sym
  1344. def strconst(self, argstr: str, i: int, delim: str) -> Tuple[int, str]:
  1345. """parse an N3 string constant delimited by delim.
  1346. return index, val
  1347. """
  1348. delim1 = delim[0]
  1349. delim2, delim3, delim4, delim5 = delim1 * 2, delim1 * 3, delim1 * 4, delim1 * 5
  1350. j = i
  1351. ustr = "" # Empty unicode string
  1352. startline = self.lines # Remember where for error messages
  1353. len_argstr = len(argstr)
  1354. while j < len_argstr:
  1355. if argstr[j] == delim1:
  1356. if delim == delim1: # done when delim is " or '
  1357. i = j + 1
  1358. return i, ustr
  1359. if (
  1360. delim == delim3
  1361. ): # done when delim is """ or ''' and, respectively ...
  1362. if argstr[j : j + 5] == delim5: # ... we have "" or '' before
  1363. i = j + 5
  1364. ustr += delim2
  1365. return i, ustr
  1366. if argstr[j : j + 4] == delim4: # ... we have " or ' before
  1367. i = j + 4
  1368. ustr += delim1
  1369. return i, ustr
  1370. if argstr[j : j + 3] == delim3: # current " or ' is part of delim
  1371. i = j + 3
  1372. return i, ustr
  1373. # we are inside of the string and current char is " or '
  1374. j += 1
  1375. ustr += delim1
  1376. continue
  1377. m = interesting.search(argstr, j) # was argstr[j:].
  1378. # Note for pos param to work, MUST be compiled ... re bug?
  1379. assert m, "Quote expected in string at ^ in %s^%s" % (
  1380. argstr[j - 20 : j],
  1381. argstr[j : j + 20],
  1382. ) # at least need a quote
  1383. i = m.start()
  1384. try:
  1385. ustr += argstr[j:i]
  1386. except UnicodeError:
  1387. err = ""
  1388. for c in argstr[j:i]:
  1389. err = err + (" %02x" % ord(c))
  1390. streason = sys.exc_info()[1].__str__()
  1391. raise BadSyntax(
  1392. self._thisDoc,
  1393. startline,
  1394. argstr,
  1395. j,
  1396. "Unicode error appending characters"
  1397. + " %s to string, because\n\t%s" % (err, streason),
  1398. )
  1399. # print "@@@ i = ",i, " j=",j, "m.end=", m.end()
  1400. ch = argstr[i]
  1401. if ch == delim1:
  1402. j = i
  1403. continue
  1404. elif ch in {'"', "'"} and ch != delim1:
  1405. ustr += ch
  1406. j = i + 1
  1407. continue
  1408. elif ch in {"\r", "\n"}:
  1409. if delim == delim1:
  1410. raise BadSyntax(
  1411. self._thisDoc,
  1412. startline,
  1413. argstr,
  1414. i,
  1415. "newline found in string literal",
  1416. )
  1417. self.lines += 1
  1418. ustr += ch
  1419. j = i + 1
  1420. self.startOfLine = j
  1421. elif ch == "\\":
  1422. j = i + 1
  1423. ch = argstr[j] # Will be empty if string ends
  1424. if not ch:
  1425. raise BadSyntax(
  1426. self._thisDoc,
  1427. startline,
  1428. argstr,
  1429. i,
  1430. "unterminated string literal (2)",
  1431. )
  1432. k = "abfrtvn\\\"'".find(ch)
  1433. if k >= 0:
  1434. uch = "\a\b\f\r\t\v\n\\\"'"[k]
  1435. ustr += uch
  1436. j += 1
  1437. elif ch == "u":
  1438. j, ch = self.uEscape(argstr, j + 1, startline)
  1439. ustr += ch
  1440. elif ch == "U":
  1441. j, ch = self.UEscape(argstr, j + 1, startline)
  1442. ustr += ch
  1443. else:
  1444. self.BadSyntax(argstr, i, "bad escape")
  1445. self.BadSyntax(argstr, i, "unterminated string literal")
  1446. def _unicodeEscape(
  1447. self,
  1448. argstr: str,
  1449. i: int,
  1450. startline: int,
  1451. reg: Pattern[str],
  1452. n: int,
  1453. prefix: str,
  1454. ) -> Tuple[int, str]:
  1455. if len(argstr) < i + n:
  1456. raise BadSyntax(
  1457. self._thisDoc, startline, argstr, i, "unterminated string literal(3)"
  1458. )
  1459. try:
  1460. return i + n, reg.sub(unicodeExpand, "\\" + prefix + argstr[i : i + n])
  1461. except Exception:
  1462. raise BadSyntax(
  1463. self._thisDoc,
  1464. startline,
  1465. argstr,
  1466. i,
  1467. "bad string literal hex escape: " + argstr[i : i + n],
  1468. )
  1469. def uEscape(self, argstr: str, i: int, startline: int) -> Tuple[int, str]:
  1470. return self._unicodeEscape(argstr, i, startline, unicodeEscape4, 4, "u")
  1471. def UEscape(self, argstr: str, i: int, startline: int) -> Tuple[int, str]:
  1472. return self._unicodeEscape(argstr, i, startline, unicodeEscape8, 8, "U")
  1473. def BadSyntax(self, argstr: str, i: int, msg: str) -> NoReturn:
  1474. raise BadSyntax(self._thisDoc, self.lines, argstr, i, msg)
  1475. # If we are going to do operators then they should generate
  1476. # [ is operator:plus of ( \1 \2 ) ]
  1477. class BadSyntax(SyntaxError): # noqa: N818
  1478. def __init__(self, uri: str, lines: int, argstr: str, i: int, why: str):
  1479. self._str = argstr.encode("utf-8") # Better go back to strings for errors
  1480. self._i = i
  1481. self._why = why
  1482. self.lines = lines
  1483. self._uri = uri
  1484. def __str__(self) -> str:
  1485. argstr = self._str
  1486. i = self._i
  1487. st = 0
  1488. if i > 60:
  1489. pre = "..."
  1490. st = i - 60
  1491. else:
  1492. pre = ""
  1493. if len(argstr) - i > 60:
  1494. post = "..."
  1495. else:
  1496. post = ""
  1497. # type error: On Python 3 formatting "b'abc'" with "%s" produces "b'abc'", not "abc"; use "%r" if this is desired behavior
  1498. return 'at line %i of <%s>:\nBad syntax (%s) at ^ in:\n"%s%s^%s%s"' % (
  1499. self.lines + 1, # type: ignore[str-bytes-safe]
  1500. self._uri,
  1501. self._why,
  1502. pre,
  1503. argstr[st:i],
  1504. argstr[i : i + 60],
  1505. post,
  1506. )
  1507. @property
  1508. def message(self) -> str:
  1509. return str(self)
  1510. ###############################################################################
  1511. class Formula:
  1512. number = 0
  1513. def __init__(self, parent: Graph):
  1514. self.uuid = uuid4().hex
  1515. self.counter = 0
  1516. Formula.number += 1
  1517. self.number = Formula.number
  1518. self.existentials: Dict[str, BNode] = {}
  1519. self.universals: Dict[str, BNode] = {}
  1520. self.quotedgraph = QuotedGraph(store=parent.store, identifier=self.id())
  1521. def __str__(self) -> str:
  1522. return "_:Formula%s" % self.number
  1523. def id(self) -> BNode:
  1524. return BNode("_:Formula%s" % self.number)
  1525. def newBlankNode(
  1526. self, uri: Optional[str] = None, why: Optional[Any] = None
  1527. ) -> BNode:
  1528. if uri is None:
  1529. self.counter += 1
  1530. bn = BNode("f%sb%s" % (self.uuid, self.counter))
  1531. else:
  1532. bn = BNode(uri.split("#").pop().replace("_", "b"))
  1533. return bn
  1534. def newUniversal(self, uri: str, why: Optional[Any] = None) -> Variable:
  1535. return Variable(uri.split("#").pop())
  1536. def declareExistential(self, x: str) -> None:
  1537. self.existentials[x] = self.newBlankNode()
  1538. def close(self) -> QuotedGraph:
  1539. return self.quotedgraph
  1540. r_hibyte = re.compile(r"([\x80-\xff])")
  1541. class RDFSink:
  1542. def __init__(self, graph: Graph):
  1543. self.rootFormula: Optional[Formula] = None
  1544. self.uuid = uuid4().hex
  1545. self.counter = 0
  1546. self.graph = graph
  1547. def newFormula(self) -> Formula:
  1548. fa = getattr(self.graph.store, "formula_aware", False)
  1549. if not fa:
  1550. raise ParserError(
  1551. "Cannot create formula parser with non-formula-aware store."
  1552. )
  1553. f = Formula(self.graph)
  1554. return f
  1555. def newGraph(self, identifier: Identifier) -> Graph:
  1556. return Graph(self.graph.store, identifier)
  1557. def newSymbol(self, *args: str) -> URIRef:
  1558. return URIRef(args[0])
  1559. def newBlankNode(
  1560. self,
  1561. arg: Optional[Union[Formula, Graph, Any]] = None,
  1562. uri: Optional[str] = None,
  1563. why: Optional[Callable[[], None]] = None,
  1564. ) -> BNode:
  1565. if isinstance(arg, Formula):
  1566. return arg.newBlankNode(uri)
  1567. elif isinstance(arg, Graph) or arg is None:
  1568. self.counter += 1
  1569. bn = BNode("n%sb%s" % (self.uuid, self.counter))
  1570. else:
  1571. bn = BNode(str(arg[0]).split("#").pop().replace("_", "b"))
  1572. return bn
  1573. def newLiteral(self, s: str, dt: Optional[URIRef], lang: Optional[str]) -> Literal:
  1574. if dt:
  1575. return Literal(s, datatype=dt)
  1576. else:
  1577. return Literal(s, lang=lang)
  1578. def newList(self, n: typing.List[Any], f: Optional[Formula]) -> IdentifiedNode:
  1579. nil = self.newSymbol("http://www.w3.org/1999/02/22-rdf-syntax-ns#nil")
  1580. if not n:
  1581. return nil
  1582. first = self.newSymbol("http://www.w3.org/1999/02/22-rdf-syntax-ns#first")
  1583. rest = self.newSymbol("http://www.w3.org/1999/02/22-rdf-syntax-ns#rest")
  1584. af = a = self.newBlankNode(f)
  1585. for ne in n[:-1]:
  1586. self.makeStatement((f, first, a, ne))
  1587. an = self.newBlankNode(f)
  1588. self.makeStatement((f, rest, a, an))
  1589. a = an
  1590. self.makeStatement((f, first, a, n[-1]))
  1591. self.makeStatement((f, rest, a, nil))
  1592. return af
  1593. def newSet(self, *args: _AnyT) -> Set[_AnyT]:
  1594. return set(args)
  1595. def setDefaultNamespace(self, *args: bytes) -> str:
  1596. return ":".join(repr(n) for n in args)
  1597. def makeStatement(
  1598. self,
  1599. quadruple: Tuple[Optional[Union[Formula, Graph]], Node, Node, Node],
  1600. why: Optional[Any] = None,
  1601. ) -> None:
  1602. f, p, s, o = quadruple
  1603. if hasattr(p, "formula"):
  1604. raise ParserError("Formula used as predicate")
  1605. # type error: Argument 1 to "normalise" of "RDFSink" has incompatible type "Union[Formula, Graph, None]"; expected "Optional[Formula]"
  1606. s = self.normalise(f, s) # type: ignore[arg-type]
  1607. p = self.normalise(f, p) # type: ignore[arg-type]
  1608. o = self.normalise(f, o) # type: ignore[arg-type]
  1609. if f == self.rootFormula:
  1610. # print s, p, o, '.'
  1611. self.graph.add((s, p, o))
  1612. elif isinstance(f, Formula):
  1613. f.quotedgraph.add((s, p, o))
  1614. else:
  1615. # type error: Item "None" of "Optional[Graph]" has no attribute "add"
  1616. f.add((s, p, o)) # type: ignore[union-attr]
  1617. # return str(quadruple)
  1618. def normalise(
  1619. self,
  1620. f: Optional[Formula],
  1621. n: Union[Tuple[int, str], bool, int, Decimal, sfloat, _AnyT],
  1622. ) -> Union[URIRef, Literal, BNode, _AnyT]:
  1623. if isinstance(n, tuple):
  1624. return URIRef(str(n[1]))
  1625. if isinstance(n, bool):
  1626. s = Literal(str(n).lower(), datatype=BOOLEAN_DATATYPE)
  1627. return s
  1628. if isinstance(n, int) or isinstance(n, long_type):
  1629. s = Literal(str(n), datatype=INTEGER_DATATYPE)
  1630. return s
  1631. if isinstance(n, Decimal):
  1632. value = str(n)
  1633. if value == "-0":
  1634. value = "0"
  1635. s = Literal(value, datatype=DECIMAL_DATATYPE)
  1636. return s
  1637. if isinstance(n, sfloat):
  1638. s = Literal(str(n), datatype=DOUBLE_DATATYPE)
  1639. return s
  1640. if isinstance(f, Formula):
  1641. if n in f.existentials:
  1642. if TYPE_CHECKING:
  1643. assert isinstance(n, URIRef)
  1644. return f.existentials[n]
  1645. # if isinstance(n, Var):
  1646. # if f.universals.has_key(n):
  1647. # return f.universals[n]
  1648. # f.universals[n] = f.newBlankNode()
  1649. # return f.universals[n]
  1650. # type error: Incompatible return value type (got "Union[int, _AnyT]", expected "Union[URIRef, Literal, BNode, _AnyT]") [return-value]
  1651. return n
  1652. def intern(self, something: _AnyT) -> _AnyT:
  1653. return something
  1654. def bind(self, pfx, uri) -> None:
  1655. pass # print pfx, ':', uri
  1656. def startDoc(self, formula: Optional[Formula]) -> None:
  1657. self.rootFormula = formula
  1658. def endDoc(self, formula: Optional[Formula]) -> None:
  1659. pass
  1660. ###################################################
  1661. #
  1662. # Utilities
  1663. #
  1664. def hexify(ustr: str) -> bytes:
  1665. """Use URL encoding to return an ASCII string
  1666. corresponding to the given UTF8 string
  1667. ```python
  1668. >>> hexify("http://example/a b")
  1669. b'http://example/a%20b'
  1670. ```
  1671. """
  1672. # s1=ustr.encode('utf-8')
  1673. s = ""
  1674. for ch in ustr: # .encode('utf-8'):
  1675. if ord(ch) > 126 or ord(ch) < 33:
  1676. ch = "%%%02X" % ord(ch)
  1677. else:
  1678. ch = "%c" % ord(ch)
  1679. s = s + ch
  1680. return s.encode("latin-1")
  1681. class TurtleParser(Parser):
  1682. """An RDFLib parser for Turtle
  1683. See http://www.w3.org/TR/turtle/
  1684. """
  1685. def __init__(self):
  1686. pass
  1687. def parse(
  1688. self,
  1689. source: InputSource,
  1690. graph: Graph,
  1691. encoding: Optional[str] = "utf-8",
  1692. turtle: bool = True,
  1693. ) -> None:
  1694. if encoding not in [None, "utf-8"]:
  1695. raise ParserError(
  1696. "N3/Turtle files are always utf-8 encoded, I was passed: %s" % encoding
  1697. )
  1698. sink = RDFSink(graph)
  1699. baseURI = graph.absolutize(source.getPublicId() or source.getSystemId() or "")
  1700. p = SinkParser(sink, baseURI=baseURI, turtle=turtle)
  1701. # N3 parser prefers str stream
  1702. stream = source.getCharacterStream()
  1703. if not stream:
  1704. stream = source.getByteStream()
  1705. p.loadStream(stream)
  1706. for prefix, namespace in p._bindings.items():
  1707. graph.bind(prefix, namespace)
  1708. class N3Parser(TurtleParser):
  1709. """An RDFLib parser for Notation3
  1710. See http://www.w3.org/DesignIssues/Notation3.html
  1711. """
  1712. def __init__(self):
  1713. pass
  1714. # type error: Signature of "parse" incompatible with supertype "TurtleParser"
  1715. def parse( # type: ignore[override]
  1716. self, source: InputSource, graph: Graph, encoding: Optional[str] = "utf-8"
  1717. ) -> None:
  1718. # we're currently being handed a Graph, not a ConjunctiveGraph
  1719. # context-aware is this implied by formula_aware
  1720. ca = getattr(graph.store, "context_aware", False)
  1721. fa = getattr(graph.store, "formula_aware", False)
  1722. if not ca:
  1723. raise ParserError("Cannot parse N3 into non-context-aware store.")
  1724. elif not fa:
  1725. raise ParserError("Cannot parse N3 into non-formula-aware store.")
  1726. conj_graph = Dataset(store=graph.store)
  1727. conj_graph.default_context = graph # TODO: CG __init__ should have a
  1728. # default_context arg
  1729. # TODO: update N3Processor so that it can use conj_graph as the sink
  1730. conj_graph.namespace_manager = graph.namespace_manager
  1731. TurtleParser.parse(self, source, conj_graph, encoding, turtle=False)