Skip to content

API Reference

Neo4jInstance

The main class for all Neo4j interactions — read queries, write queries, bulk data loading, and graph EDA.

Neo4jInstance

Class use to handle Neo4j requests.

Methods:

Name Description
execute_read_query

kwargs Optional[Dict[str, Any]]) -> DataFrame Execute a read query to a specific database.

execute_write_queries

kwargs: Optional[Dict[str, Any]]) -> None Execute a list of write queries to a specific database.

execute_write_query

kwargs: Optional[Dict[str, Any]]) -> None Execute a write query to a specific database.

get_query_visualization

parameters: Optional[Dict[str, Any]]) -> VisualizationGraph Execute a read query and return the result as a neo4j-viz graph visualization.

execute_write_query_with_data

database: Optional[str] = None, partitions: Optional[int] = 1, parallel: Optional[bool] = False, workers: Optional[int] = None, parameters: Optional[Dict[str, Any]] ) -> Dict[str, int] Execute a write query using data on a DataFrame.

execute_write_queries_with_data

database: Optional[str] = None, partitions: Optional[int] = 1, parallel: Optional[bool] = False, workers: Optional[int] = None, parameters: Optional[Dict[str, Any]] ) -> Dict[str, int] Execute a list of write queries using data on a DataFrame.

Source code in pyneoinstance/database/neo4jdbms.py
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
class Neo4jInstance:
    """Class use to handle Neo4j requests.

        Methods
        -------
        execute_read_query(query: str,database Optional[str],
                           kwargs Optional[Dict[str, Any]]) -> DataFrame
            Execute a read query to a specific database.
        execute_write_queries(queries: List[str], database: Optional[str],
                            kwargs: Optional[Dict[str, Any]]) -> None
            Execute a list of write queries to a specific database.
        execute_write_query(query: str, database: Optional[str],
                            kwargs: Optional[Dict[str, Any]]) -> None
            Execute a write query to a specific database.
        get_query_visualization(query: str, database: Optional[str],
                               parameters: Optional[Dict[str, Any]]) -> VisualizationGraph
            Execute a read query and return the result as a neo4j-viz graph visualization.
        execute_write_query_with_data(self, query: str, data: DataFrame,
                                      database: Optional[str] = None,
                                      partitions: Optional[int] = 1,
                                      parallel: Optional[bool] = False,
                                      workers: Optional[int] = None,
                                      parameters: Optional[Dict[str, Any]]
                                      ) -> Dict[str, int]
            Execute a write query using data on a DataFrame.
        execute_write_queries_with_data(self, query: List[str], data: DataFrame,
                                        database: Optional[str] = None,
                                        partitions: Optional[int] = 1,
                                        parallel: Optional[bool] = False,
                                        workers: Optional[int] = None,
                                        parameters: Optional[Dict[str, Any]]
                                        ) -> Dict[str, int]
            Execute a list of write queries using data on a DataFrame.
    """
    def __init__(self, uri: str, user: str, password: str,
                 verbose: bool = False, **kwargs: Optional[Dict[str, Any]]) -> None:
        """Class constructor.

        Parameters
        ----------
        uri : str
            The Uniform Resource Identifier of the database.
        user : str
            The Neo4j username.
        password : str
            The unencrypted Neo4j user password.
        kwargs : Dict[str, Any], optional
            Extra arguments, more notably encrypted (bool)

        Raises
        ------
        AuthError
            When the authentication to Neo4j fails.
        ServiceUnavailable
            When the Neo4j instance is not available.
        ConfigurationError
            When the Neo4j URI is not valid.
        """

        self.neo_info = {}
        self.neo_info['uri'] = uri
        self.neo_info['user'] = user
        self.neo_info['password'] = password
        self.neo_info['encrypted'] = kwargs.get('encrypted') or ''
        logging_type = logging.INFO if verbose else logging.WARNING
        self.__logger = get_logger('pyneoinstance', logging_type)
        if not self.__logger.handlers:
            stream_handler = logging.StreamHandler()
            stream_handler.setFormatter(logging.Formatter('%(levelname)s - %(message)s'))
            self.__logger.addHandler(stream_handler)
        self._driver = _get_driver(self.neo_info)

    def close(self):
        self._driver.close()

    def __enter__(self):
        return self

    def __exit__(self, exc_type, exc_val, exc_tb):
        self.close()

    _apoc_error_msg = "APOC Library not detected. Please install it before proceeding."

    def _execute_read(self, query: str,
                      database: Optional[str] = None,
                      error_msg: Optional[str] = None) -> DataFrame:
        with _get_session(self._driver, database) as session:
            try:
                return session.execute_read(
                    _read_transaction_function, query=query)
            except ServiceUnavailable as exception:
                raise ServiceUnavailable() from exception
            except ClientError as exception:
                raise ClientError(error_msg) from exception

    def execute_read_query(self, query: str,
                           database: Optional[str] = None,
                           parameters: Optional[Dict[str, Any]] = None
                          ) -> DataFrame:
        """Execute a read transaction to a specific database.

        Parameters
        ----------
        query : str
            String containing the Cypher query to execute.
        database : str, optional
            Name of the Neo4j database of which to execute the
            transaction. If not provided the default database is going to
            be use.
        parameters: Dict[str, Any], optional
            Extra arguments containing optional cypher parameters.

        Return
        ------
        DataFrame
            Pandas DataFrame with the results of the transaction.

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When there is a Cypher syntax error or datatype error.
    """
        params = parameters or {}
        with _get_session(self._driver, database) as session:
            try:
                if _IMPLICIT_TX_RE.search(query):
                    # 'CALL { ... } IN TRANSACTIONS' is only valid in an implicit
                    # (auto-commit) transaction, so it can't go through execute_read.
                    # session.run isn't retried by the driver, so retry manually.
                    result = _run_autocommit_with_retry(
                        session,
                        lambda r: DataFrame(r.values(), columns=r.keys()),
                        query, **params)
                else:
                    result = session.execute_read(
                        _read_transaction_function,query=query,**params)
            except ServiceUnavailable as exception:
                raise ServiceUnavailable() from exception
            except ClientError as exception:
                raise ClientError(str(exception)) from exception
        return result

    def get_query_visualization(self, query: str,
                                database: Optional[str] = None,
                                parameters: Optional[Dict[str, Any]] = None
                               ) -> VisualizationGraph:
        """Execute a read query and return the result as a graph visualization.

            Parameters
            ----------
            query : str
                Cypher query returning nodes and/or relationships.
            database : str, optional
                Name of the Neo4j database on which to execute the query.
                If not provided the default database is used.
            parameters : Dict[str, Any], optional
                Extra arguments containing optional Cypher parameters.

            Returns
            -------
            VisualizationGraph
                neo4j-viz VisualizationGraph object. Node captions are derived
                from node labels; relationship captions from relationship type.
                In a Jupyter notebook call ``vg.render()`` to display it inline.
                To save as a standalone HTML file, wrap the fragment in a full document::

                    fragment = vg.render().data
                    with open('graph.html', 'w') as f:
                        f.write(f'<!DOCTYPE html><html><body>{fragment}</body></html>')

            Raises
            ------
            ServiceUnavailable
                When the Neo4j instance is not available.
            ClientError
                When there is a Cypher syntax error or datatype error.
        """
        params = parameters or {}
        with _get_session(self._driver, database) as session:
            try:
                if _IMPLICIT_TX_RE.search(query):
                    # 'CALL { ... } IN TRANSACTIONS' is only valid in an implicit
                    # (auto-commit) transaction, so it can't go through execute_read.
                    # session.run isn't retried by the driver, so retry manually.
                    def _consume_graph(result):
                        list(result)  # consume records so .graph() is populated
                        return result.graph()
                    graph = _run_autocommit_with_retry(
                        session, _consume_graph, query, **params)
                else:
                    graph = session.execute_read(
                        _graph_transaction_function, query=query, **params)
            except ServiceUnavailable as exception:
                raise ServiceUnavailable() from exception
            except ClientError as exception:
                raise ClientError(str(exception)) from exception
        return from_neo4j(graph)

    def execute_write_queries(self, queries: List[str],
                            database: Optional[str] = None,
                            parameters: Optional[Dict[str, Any]] = None
                           ) -> Dict[str, int]:
        """Execute a write query to a specific database.

            Parameters
            ----------
            queries : List[str]
                List of Cypher queries to execute.
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.
            parameters : Dict[str, Any], optional
                Extra arguments containing optional cypher parameters.

            Returns
            -------
            Dictionary
                Python dictionary containing Neo4j write counts (nodes_created,
                labels_added, properties_set, ect.)

            Raises
            ------
            ServiceUnavailable
                When the Neo4j instance is not available.
            ClientError
                When their is a Cypher syntax error or datatype error.
        """
        params = parameters or {}
        results = defaultdict(int)
        with _get_session(self._driver, database) as session:
            for query in queries:
                result = self._execute_write(session, query, parameters=params)
                for key, value in result.items():
                    if key != '_contains_updates':
                        results[key] += value
        return dict(results)

    def execute_write_query(self, query: str,
                            database: Optional[str] = None,
                            parameters: Optional[Dict[str, Any]] = None
                           ) -> Dict[str, Any]:
        """Execute a write query to a specific database.

            Parameters
            ----------
            query : str
                Cypher query to execute.
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.
            parameters : Dict[str, Any], optional
                Extra arguments containing optional cypher parameters.

            Returns
            -------
            Dictionary
                Python dictionary containing Neo4j write counts (nodes_created,
                labels_added, properties_set, ect.)

            Raises
            ------
            ServiceUnavailable
                When the Neo4j instance is not available.
            ClientError
                When their is a Cypher syntax error or datatype error.
        """
        result = self.execute_write_queries([query], database, parameters)
        return  result

    def execute_write_queries_with_data(self, queries: List[str],
                                        data: DataFrame,
                                        database: Optional[str] = None,
                                        batchSize: int = 100_000,
                                        parallel: bool = False,
                                        workers: Optional[int] = None,
                                        parameters: Optional[Dict[str, Any]] = None
                                       ) -> Dict[str, int]:
        """Execute a list of write queries using data to update a specific database.

            Parameters
            ----------
            queries : List[str]
                List of strings constaining the Cyphrer queries to execute.
            data : DataFrame
                Pandas DataFrame containing data to process.
            database : str, optional
                Name of the Neo4j database of which to execute the transactions.
                If not provided the default database is going to be use.
            batchSize : int, optional
                The number of records per batch or partitions of the data frame.
            parallel : bool, optional
                Wheather to execute the load in parallel.
            workers : int, optional
                Number of processes to spawn to load the data.
            parameters : Dict[str, Any], optional
                Extra arguments containing optional cypher parameters.

            Returns
            -------
            Dictionary
                Python dictionary containing Neo4j write counts (nodes_created,
                labels_added, properties_set, ect.)

            Raises
            ------
            ServiceUnavailable
                When the Neo4j instance is not available.
            ClientError
                When their is a Cypher syntax error or datatype error.
            ValueError
                When the number of batch sizes to split the DataFrame on
                is larger than the number of rows in it.
        """
        row_num = data.shape[0]
        batchSize = min(batchSize, row_num)
        self.__logger.info(f'Partitioning the data in batches of size {batchSize:,.0f}')
        chunks = get_batches(data, batchSize)
        partitions = len(chunks)
        parallel = parallel if partitions > 1 else False
        params = parameters or {}

        for query in queries:
            col_diff = get_columns_diff(query, data.columns)
            if col_diff:
                self.__logger.warning(f'These columns are not in your data: {col_diff}')

        results = []
        if parallel:
            workers_num = workers if (workers and workers > 0) else os.cpu_count()
            self.__logger.info(
                f'Loading {partitions:,.0f} data chunks using {workers_num} thread(s)')

            with ThreadPoolExecutor(max_workers=workers_num) as executor:
                for query in queries:
                    def run_chunk(chunk_indices, q=query):
                        rows = data.loc[chunk_indices].to_dict('records')
                        with _get_session(self._driver, database) as session:
                            return self._execute_write(session, q, rows, params)
                    results.extend(executor.map(run_chunk, chunks))
        else:
            self.__logger.info(f'Loading {partitions:,.0f} data chunks sequentially')
            with _get_session(self._driver, database) as session:
                for query in queries:
                    for chunk_indices in chunks:
                        rows = data.loc[chunk_indices].to_dict('records')
                        results.append(self._execute_write(session, query, rows, params))

        results_agg = defaultdict(int)
        for result in results:
            for key, value in result.items():
                if key != '_contains_updates':
                    results_agg[key] += value
        return dict(results_agg)

    def execute_write_query_with_data(self,
                                      query: str, data: DataFrame,
                                      database: Optional[str] = None,
                                      batchSize: Optional[int] = 100_000,
                                      parallel: Optional[bool] = False,
                                      workers: Optional[int] = None,
                                      parameters: Optional[Dict[str, Any]] = None
                                     ) -> Dict[str, int]:
        """Execute a write query with data to update a specific database.

            Parameters
            ----------
            query : str
                Cypher query to execute.
            data : DataFrame
                Pandas DataFrame containing data to process.
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.
            batchSize : int, optional
                The number of records per batch or partitions of the data frame.
            parallel : bool, optional
                Wheather to execute the load in parallel.
            workers : int, optional
                Number of processes to spawn to load the data.
            parameters : Dict[str, Any], optional
                Extra arguments containing optional cypher parameters.

            Returns
            -------
            Dictionary
                Python dictionary containing Neo4j write counts (nodes_created,
                labels_added, properties_set, ect.)

            Raises
            ------
            ServiceUnavailable
                When the Neo4j instance is not available.
            ClientError
                When their is a Cypher syntax error or datatype error.
            ValueError
                When the number of batch sizes to split the DataFrame on
                is larger than the number of rows in it.
        """
        result = self.execute_write_queries_with_data(
            [query], data, database, batchSize, parallel, workers, parameters)
        return result

    def get_node_label_freq(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the graph node label frequency.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing the frequency and relative
                frequency of node labels.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            MATCH(n)
            WITH count(*) AS nodeCount
            CALL db.labels() YIELD label
            CALL apoc.cypher.run('MATCH (:`'+label+'`) RETURN count(*) as freq',{}) YIELD value
            WITH nodeCount,label,value.freq AS freq
            WITH *, 10^3 AS scaleFactor, toFloat(freq)/toFloat(nodeCount) AS relFreq
            RETURN label AS nodeLabel,
                freq AS frequency,
                round(relFreq*scaleFactor)/scaleFactor AS relativeFrequency
            ORDER BY freq DESC
        """
        return self._execute_read(query, database, self._apoc_error_msg)

    def get_node_multilabel_freq(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the graph node multi-label frequency.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing the frequency and relative
                frequency of node with multiple labels.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            MATCH (n)
            WITH labels(n) as nodeLabels
            WHERE size(nodeLabels)>1
            RETURN nodeLabels, count(*) as frequency
        """
        return self._execute_read(query, database)

    def get_rela_type_freq(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the graph relationship type frequency.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing the frequency and relative
                frequency of relationship types.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            MATCH()-[]->()
            WITH count(*) AS relCount
            CALL db.relationshipTypes() YIELD relationshipType as type
            CALL apoc.cypher.run('MATCH ()-[:`'+type+'`]->() RETURN count(*) as freq',{})
            YIELD value
            WITH type AS relationshipType, value.freq AS freq,relCount
            WITH *,3 AS presicion
            WITH *, 10^presicion AS factor,toFloat(freq)/toFloat(relCount) as relFreq
            RETURN relationshipType, freq AS frequency,
            round(relFreq*factor)/factor AS relativeFrequency
            ORDER BY freq DESC;
        """
        return self._execute_read(query, database, self._apoc_error_msg)

    def get_properties(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the node and relationship properties.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing information about all nodes
                and relationships properties.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            CALL apoc.meta.data() YIELD label,property,type,elementType
            WHERE type<>'RELATIONSHIP'
            RETURN elementType,label,property,type
            ORDER BY elementType,label,property;
        """
        return self._execute_read(query, database, self._apoc_error_msg)

    def get_constraints(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the constraints in the graph.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing information about all the
                constraints.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            SHOW CONSTRAINTS
        """
        return self._execute_read(query, database, self._apoc_error_msg)

    def get_indexes(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the indexes in the graph.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing information about all the
                indexes.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            SHOW INDEX
        """
        return self._execute_read(query, database, self._apoc_error_msg)

    def get_rela_source_target_freq(self, database: Optional[str] = None) -> DataFrame:
        """Use to obtain the relationships source and target frequency.

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            DataFrame
                Pandas DataFrame object containing the frequency of
                relationships cource and target frequency.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            CALL apoc.meta.stats() YIELD relTypes
        """
        stat_dict = self._execute_read(query, database, self._apoc_error_msg).iloc[0,0]
        stats_dicts = []
        for key, value in stat_dict.items():
            nodes = _NODE_LABEL_RE.findall(key)
            rel_matches = _REL_TYPE_RE.findall(key)
            if len(nodes) < 2 or not rel_matches:
                continue
            info = {'sourceLabel': nodes[0], 'relationshipType': rel_matches[0],
                    'targetLabel': nodes[1], 'frequency': value}
            stats_dicts.append(info)
        stats_df = DataFrame(stats_dicts)
        stats_df.sort_values(by=['relationshipType','sourceLabel','targetLabel'],inplace=True)
        stats_df.reset_index(drop=True, inplace=True)
        return stats_df

    def get_schema_visualization(self, database: Optional[str] = None) -> Network:
        """Use to visualize the graph data model (schema).

            Parameters
            ----------
            database : str, optional
                Name of the Neo4j database of which to execute the transaction.
                If not provided the default database is going to be use.

            Returns
            -------
            Network
                Pyvis Network object. Render with ``network.write_html('schema.html')``.

            Raises
            ------
            ClientError
                When APOC Library is not install correctly in your Neo4j
                deployment.
        """

        query = """
            CALL apoc.meta.schema() YIELD value
        """
        schema = self._execute_read(query, database, self._apoc_error_msg).iloc[0,0]
        network = Network(cdn_resources="remote", directed=True,
                          filter_menu=True, height="800px", width="100%")
        options = """
        const options = {
            "physics": {
                "barnesHut": {
                  "gravitationalConstant": -13950,
                  "centralGravity": 6.15,
                  "springLength": 160,
                  "damping": 0.23
                },
                "minVelocity": 0.75
              }
            }
        """
        network.set_options(options)
        nodes = []
        relationships = set()
        for key in schema.keys():
            if schema[key]['type']=='node':
                nodes.append(key)
        network.add_nodes(nodes)
        for node in nodes:
            for rel in schema[node]['relationships'].keys():
                title = rel
                if schema[node]['relationships'][rel]['direction']=='in':
                    target = node
                    for label in schema[node]['relationships'][rel]['labels']:
                        source = label
                else:
                    source = node
                    for label in schema[node]['relationships'][rel]['labels']:
                        target = label
                relationships.add(source + "|" + title + "|" + target)
        for relationship in relationships:
            rela_parts = relationship.split("|")
            network.add_edge(rela_parts[0],rela_parts[2],title=rela_parts[1])
        return network

    def _execute_write(self, session, query,
                       rows: Optional[Dict[str, Any]] = None,
                       parameters: Optional[Dict[str, Any]] = None
                      ):
        params = parameters or {}
        kwargs = dict(params)
        if rows is not None:
            kwargs['rows'] = rows
        try:
            if _IMPLICIT_TX_RE.search(query):
                # 'CALL { ... } IN TRANSACTIONS' is only valid in an implicit
                # (auto-commit) transaction, so it can't go through execute_write.
                # session.run isn't retried by the driver, so retry manually.
                results = _run_autocommit_with_retry(
                    session, lambda r: r.consume().counters.__dict__,
                    query, **kwargs)
            else:
                results = session.execute_write(
                    _write_transaction_function, query,
                    **kwargs).counters.__dict__
        except ServiceUnavailable as exception:
            raise ServiceUnavailable() from exception
        except ClientError as exception:
            raise ClientError(str(exception)) from exception
        return results

__init__

__init__(uri: str, user: str, password: str, verbose: bool = False, **kwargs: Optional[Dict[str, Any]]) -> None

Class constructor.

Parameters:

Name Type Description Default
uri str

The Uniform Resource Identifier of the database.

required
user str

The Neo4j username.

required
password str

The unencrypted Neo4j user password.

required
kwargs Dict[str, Any]

Extra arguments, more notably encrypted (bool)

{}

Raises:

Type Description
AuthError

When the authentication to Neo4j fails.

ServiceUnavailable

When the Neo4j instance is not available.

ConfigurationError

When the Neo4j URI is not valid.

Source code in pyneoinstance/database/neo4jdbms.py
def __init__(self, uri: str, user: str, password: str,
             verbose: bool = False, **kwargs: Optional[Dict[str, Any]]) -> None:
    """Class constructor.

    Parameters
    ----------
    uri : str
        The Uniform Resource Identifier of the database.
    user : str
        The Neo4j username.
    password : str
        The unencrypted Neo4j user password.
    kwargs : Dict[str, Any], optional
        Extra arguments, more notably encrypted (bool)

    Raises
    ------
    AuthError
        When the authentication to Neo4j fails.
    ServiceUnavailable
        When the Neo4j instance is not available.
    ConfigurationError
        When the Neo4j URI is not valid.
    """

    self.neo_info = {}
    self.neo_info['uri'] = uri
    self.neo_info['user'] = user
    self.neo_info['password'] = password
    self.neo_info['encrypted'] = kwargs.get('encrypted') or ''
    logging_type = logging.INFO if verbose else logging.WARNING
    self.__logger = get_logger('pyneoinstance', logging_type)
    if not self.__logger.handlers:
        stream_handler = logging.StreamHandler()
        stream_handler.setFormatter(logging.Formatter('%(levelname)s - %(message)s'))
        self.__logger.addHandler(stream_handler)
    self._driver = _get_driver(self.neo_info)

execute_read_query

execute_read_query(query: str, database: Optional[str] = None, parameters: Optional[Dict[str, Any]] = None) -> DataFrame

Execute a read transaction to a specific database.

Parameters:

Name Type Description Default
query str

String containing the Cypher query to execute.

required
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None
parameters Optional[Dict[str, Any]]

Extra arguments containing optional cypher parameters.

None
Return

DataFrame Pandas DataFrame with the results of the transaction.

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When there is a Cypher syntax error or datatype error.

Source code in pyneoinstance/database/neo4jdbms.py
def execute_read_query(self, query: str,
                       database: Optional[str] = None,
                       parameters: Optional[Dict[str, Any]] = None
                      ) -> DataFrame:
    """Execute a read transaction to a specific database.

    Parameters
    ----------
    query : str
        String containing the Cypher query to execute.
    database : str, optional
        Name of the Neo4j database of which to execute the
        transaction. If not provided the default database is going to
        be use.
    parameters: Dict[str, Any], optional
        Extra arguments containing optional cypher parameters.

    Return
    ------
    DataFrame
        Pandas DataFrame with the results of the transaction.

    Raises
    ------
    ServiceUnavailable
        When the Neo4j instance is not available.
    ClientError
        When there is a Cypher syntax error or datatype error.
"""
    params = parameters or {}
    with _get_session(self._driver, database) as session:
        try:
            if _IMPLICIT_TX_RE.search(query):
                # 'CALL { ... } IN TRANSACTIONS' is only valid in an implicit
                # (auto-commit) transaction, so it can't go through execute_read.
                # session.run isn't retried by the driver, so retry manually.
                result = _run_autocommit_with_retry(
                    session,
                    lambda r: DataFrame(r.values(), columns=r.keys()),
                    query, **params)
            else:
                result = session.execute_read(
                    _read_transaction_function,query=query,**params)
        except ServiceUnavailable as exception:
            raise ServiceUnavailable() from exception
        except ClientError as exception:
            raise ClientError(str(exception)) from exception
    return result

get_query_visualization

get_query_visualization(query: str, database: Optional[str] = None, parameters: Optional[Dict[str, Any]] = None) -> VisualizationGraph

Execute a read query and return the result as a graph visualization.

Parameters:

Name Type Description Default
query str

Cypher query returning nodes and/or relationships.

required
database str

Name of the Neo4j database on which to execute the query. If not provided the default database is used.

None
parameters Dict[str, Any]

Extra arguments containing optional Cypher parameters.

None

Returns:

Type Description
VisualizationGraph

neo4j-viz VisualizationGraph object. Node captions are derived from node labels; relationship captions from relationship type. In a Jupyter notebook call vg.render() to display it inline. To save as a standalone HTML file, wrap the fragment in a full document::

fragment = vg.render().data
with open('graph.html', 'w') as f:
    f.write(f'<!DOCTYPE html><html><body>{fragment}</body></html>')

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When there is a Cypher syntax error or datatype error.

Source code in pyneoinstance/database/neo4jdbms.py
def get_query_visualization(self, query: str,
                            database: Optional[str] = None,
                            parameters: Optional[Dict[str, Any]] = None
                           ) -> VisualizationGraph:
    """Execute a read query and return the result as a graph visualization.

        Parameters
        ----------
        query : str
            Cypher query returning nodes and/or relationships.
        database : str, optional
            Name of the Neo4j database on which to execute the query.
            If not provided the default database is used.
        parameters : Dict[str, Any], optional
            Extra arguments containing optional Cypher parameters.

        Returns
        -------
        VisualizationGraph
            neo4j-viz VisualizationGraph object. Node captions are derived
            from node labels; relationship captions from relationship type.
            In a Jupyter notebook call ``vg.render()`` to display it inline.
            To save as a standalone HTML file, wrap the fragment in a full document::

                fragment = vg.render().data
                with open('graph.html', 'w') as f:
                    f.write(f'<!DOCTYPE html><html><body>{fragment}</body></html>')

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When there is a Cypher syntax error or datatype error.
    """
    params = parameters or {}
    with _get_session(self._driver, database) as session:
        try:
            if _IMPLICIT_TX_RE.search(query):
                # 'CALL { ... } IN TRANSACTIONS' is only valid in an implicit
                # (auto-commit) transaction, so it can't go through execute_read.
                # session.run isn't retried by the driver, so retry manually.
                def _consume_graph(result):
                    list(result)  # consume records so .graph() is populated
                    return result.graph()
                graph = _run_autocommit_with_retry(
                    session, _consume_graph, query, **params)
            else:
                graph = session.execute_read(
                    _graph_transaction_function, query=query, **params)
        except ServiceUnavailable as exception:
            raise ServiceUnavailable() from exception
        except ClientError as exception:
            raise ClientError(str(exception)) from exception
    return from_neo4j(graph)

execute_write_queries

execute_write_queries(queries: List[str], database: Optional[str] = None, parameters: Optional[Dict[str, Any]] = None) -> Dict[str, int]

Execute a write query to a specific database.

Parameters:

Name Type Description Default
queries List[str]

List of Cypher queries to execute.

required
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None
parameters Dict[str, Any]

Extra arguments containing optional cypher parameters.

None

Returns:

Type Description
Dictionary

Python dictionary containing Neo4j write counts (nodes_created, labels_added, properties_set, ect.)

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When their is a Cypher syntax error or datatype error.

Source code in pyneoinstance/database/neo4jdbms.py
def execute_write_queries(self, queries: List[str],
                        database: Optional[str] = None,
                        parameters: Optional[Dict[str, Any]] = None
                       ) -> Dict[str, int]:
    """Execute a write query to a specific database.

        Parameters
        ----------
        queries : List[str]
            List of Cypher queries to execute.
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.
        parameters : Dict[str, Any], optional
            Extra arguments containing optional cypher parameters.

        Returns
        -------
        Dictionary
            Python dictionary containing Neo4j write counts (nodes_created,
            labels_added, properties_set, ect.)

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When their is a Cypher syntax error or datatype error.
    """
    params = parameters or {}
    results = defaultdict(int)
    with _get_session(self._driver, database) as session:
        for query in queries:
            result = self._execute_write(session, query, parameters=params)
            for key, value in result.items():
                if key != '_contains_updates':
                    results[key] += value
    return dict(results)

execute_write_query

execute_write_query(query: str, database: Optional[str] = None, parameters: Optional[Dict[str, Any]] = None) -> Dict[str, Any]

Execute a write query to a specific database.

Parameters:

Name Type Description Default
query str

Cypher query to execute.

required
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None
parameters Dict[str, Any]

Extra arguments containing optional cypher parameters.

None

Returns:

Type Description
Dictionary

Python dictionary containing Neo4j write counts (nodes_created, labels_added, properties_set, ect.)

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When their is a Cypher syntax error or datatype error.

Source code in pyneoinstance/database/neo4jdbms.py
def execute_write_query(self, query: str,
                        database: Optional[str] = None,
                        parameters: Optional[Dict[str, Any]] = None
                       ) -> Dict[str, Any]:
    """Execute a write query to a specific database.

        Parameters
        ----------
        query : str
            Cypher query to execute.
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.
        parameters : Dict[str, Any], optional
            Extra arguments containing optional cypher parameters.

        Returns
        -------
        Dictionary
            Python dictionary containing Neo4j write counts (nodes_created,
            labels_added, properties_set, ect.)

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When their is a Cypher syntax error or datatype error.
    """
    result = self.execute_write_queries([query], database, parameters)
    return  result

execute_write_queries_with_data

execute_write_queries_with_data(queries: List[str], data: DataFrame, database: Optional[str] = None, batchSize: int = 100000, parallel: bool = False, workers: Optional[int] = None, parameters: Optional[Dict[str, Any]] = None) -> Dict[str, int]

Execute a list of write queries using data to update a specific database.

Parameters:

Name Type Description Default
queries List[str]

List of strings constaining the Cyphrer queries to execute.

required
data DataFrame

Pandas DataFrame containing data to process.

required
database str

Name of the Neo4j database of which to execute the transactions. If not provided the default database is going to be use.

None
batchSize int

The number of records per batch or partitions of the data frame.

100000
parallel bool

Wheather to execute the load in parallel.

False
workers int

Number of processes to spawn to load the data.

None
parameters Dict[str, Any]

Extra arguments containing optional cypher parameters.

None

Returns:

Type Description
Dictionary

Python dictionary containing Neo4j write counts (nodes_created, labels_added, properties_set, ect.)

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When their is a Cypher syntax error or datatype error.

ValueError

When the number of batch sizes to split the DataFrame on is larger than the number of rows in it.

Source code in pyneoinstance/database/neo4jdbms.py
def execute_write_queries_with_data(self, queries: List[str],
                                    data: DataFrame,
                                    database: Optional[str] = None,
                                    batchSize: int = 100_000,
                                    parallel: bool = False,
                                    workers: Optional[int] = None,
                                    parameters: Optional[Dict[str, Any]] = None
                                   ) -> Dict[str, int]:
    """Execute a list of write queries using data to update a specific database.

        Parameters
        ----------
        queries : List[str]
            List of strings constaining the Cyphrer queries to execute.
        data : DataFrame
            Pandas DataFrame containing data to process.
        database : str, optional
            Name of the Neo4j database of which to execute the transactions.
            If not provided the default database is going to be use.
        batchSize : int, optional
            The number of records per batch or partitions of the data frame.
        parallel : bool, optional
            Wheather to execute the load in parallel.
        workers : int, optional
            Number of processes to spawn to load the data.
        parameters : Dict[str, Any], optional
            Extra arguments containing optional cypher parameters.

        Returns
        -------
        Dictionary
            Python dictionary containing Neo4j write counts (nodes_created,
            labels_added, properties_set, ect.)

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When their is a Cypher syntax error or datatype error.
        ValueError
            When the number of batch sizes to split the DataFrame on
            is larger than the number of rows in it.
    """
    row_num = data.shape[0]
    batchSize = min(batchSize, row_num)
    self.__logger.info(f'Partitioning the data in batches of size {batchSize:,.0f}')
    chunks = get_batches(data, batchSize)
    partitions = len(chunks)
    parallel = parallel if partitions > 1 else False
    params = parameters or {}

    for query in queries:
        col_diff = get_columns_diff(query, data.columns)
        if col_diff:
            self.__logger.warning(f'These columns are not in your data: {col_diff}')

    results = []
    if parallel:
        workers_num = workers if (workers and workers > 0) else os.cpu_count()
        self.__logger.info(
            f'Loading {partitions:,.0f} data chunks using {workers_num} thread(s)')

        with ThreadPoolExecutor(max_workers=workers_num) as executor:
            for query in queries:
                def run_chunk(chunk_indices, q=query):
                    rows = data.loc[chunk_indices].to_dict('records')
                    with _get_session(self._driver, database) as session:
                        return self._execute_write(session, q, rows, params)
                results.extend(executor.map(run_chunk, chunks))
    else:
        self.__logger.info(f'Loading {partitions:,.0f} data chunks sequentially')
        with _get_session(self._driver, database) as session:
            for query in queries:
                for chunk_indices in chunks:
                    rows = data.loc[chunk_indices].to_dict('records')
                    results.append(self._execute_write(session, query, rows, params))

    results_agg = defaultdict(int)
    for result in results:
        for key, value in result.items():
            if key != '_contains_updates':
                results_agg[key] += value
    return dict(results_agg)

execute_write_query_with_data

execute_write_query_with_data(query: str, data: DataFrame, database: Optional[str] = None, batchSize: Optional[int] = 100000, parallel: Optional[bool] = False, workers: Optional[int] = None, parameters: Optional[Dict[str, Any]] = None) -> Dict[str, int]

Execute a write query with data to update a specific database.

Parameters:

Name Type Description Default
query str

Cypher query to execute.

required
data DataFrame

Pandas DataFrame containing data to process.

required
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None
batchSize int

The number of records per batch or partitions of the data frame.

100000
parallel bool

Wheather to execute the load in parallel.

False
workers int

Number of processes to spawn to load the data.

None
parameters Dict[str, Any]

Extra arguments containing optional cypher parameters.

None

Returns:

Type Description
Dictionary

Python dictionary containing Neo4j write counts (nodes_created, labels_added, properties_set, ect.)

Raises:

Type Description
ServiceUnavailable

When the Neo4j instance is not available.

ClientError

When their is a Cypher syntax error or datatype error.

ValueError

When the number of batch sizes to split the DataFrame on is larger than the number of rows in it.

Source code in pyneoinstance/database/neo4jdbms.py
def execute_write_query_with_data(self,
                                  query: str, data: DataFrame,
                                  database: Optional[str] = None,
                                  batchSize: Optional[int] = 100_000,
                                  parallel: Optional[bool] = False,
                                  workers: Optional[int] = None,
                                  parameters: Optional[Dict[str, Any]] = None
                                 ) -> Dict[str, int]:
    """Execute a write query with data to update a specific database.

        Parameters
        ----------
        query : str
            Cypher query to execute.
        data : DataFrame
            Pandas DataFrame containing data to process.
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.
        batchSize : int, optional
            The number of records per batch or partitions of the data frame.
        parallel : bool, optional
            Wheather to execute the load in parallel.
        workers : int, optional
            Number of processes to spawn to load the data.
        parameters : Dict[str, Any], optional
            Extra arguments containing optional cypher parameters.

        Returns
        -------
        Dictionary
            Python dictionary containing Neo4j write counts (nodes_created,
            labels_added, properties_set, ect.)

        Raises
        ------
        ServiceUnavailable
            When the Neo4j instance is not available.
        ClientError
            When their is a Cypher syntax error or datatype error.
        ValueError
            When the number of batch sizes to split the DataFrame on
            is larger than the number of rows in it.
    """
    result = self.execute_write_queries_with_data(
        [query], data, database, batchSize, parallel, workers, parameters)
    return result

get_node_label_freq

get_node_label_freq(database: Optional[str] = None) -> DataFrame

Use to obtain the graph node label frequency.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing the frequency and relative frequency of node labels.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_node_label_freq(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the graph node label frequency.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing the frequency and relative
            frequency of node labels.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        MATCH(n)
        WITH count(*) AS nodeCount
        CALL db.labels() YIELD label
        CALL apoc.cypher.run('MATCH (:`'+label+'`) RETURN count(*) as freq',{}) YIELD value
        WITH nodeCount,label,value.freq AS freq
        WITH *, 10^3 AS scaleFactor, toFloat(freq)/toFloat(nodeCount) AS relFreq
        RETURN label AS nodeLabel,
            freq AS frequency,
            round(relFreq*scaleFactor)/scaleFactor AS relativeFrequency
        ORDER BY freq DESC
    """
    return self._execute_read(query, database, self._apoc_error_msg)

get_node_multilabel_freq

get_node_multilabel_freq(database: Optional[str] = None) -> DataFrame

Use to obtain the graph node multi-label frequency.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing the frequency and relative frequency of node with multiple labels.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_node_multilabel_freq(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the graph node multi-label frequency.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing the frequency and relative
            frequency of node with multiple labels.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        MATCH (n)
        WITH labels(n) as nodeLabels
        WHERE size(nodeLabels)>1
        RETURN nodeLabels, count(*) as frequency
    """
    return self._execute_read(query, database)

get_rela_type_freq

get_rela_type_freq(database: Optional[str] = None) -> DataFrame

Use to obtain the graph relationship type frequency.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing the frequency and relative frequency of relationship types.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_rela_type_freq(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the graph relationship type frequency.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing the frequency and relative
            frequency of relationship types.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        MATCH()-[]->()
        WITH count(*) AS relCount
        CALL db.relationshipTypes() YIELD relationshipType as type
        CALL apoc.cypher.run('MATCH ()-[:`'+type+'`]->() RETURN count(*) as freq',{})
        YIELD value
        WITH type AS relationshipType, value.freq AS freq,relCount
        WITH *,3 AS presicion
        WITH *, 10^presicion AS factor,toFloat(freq)/toFloat(relCount) as relFreq
        RETURN relationshipType, freq AS frequency,
        round(relFreq*factor)/factor AS relativeFrequency
        ORDER BY freq DESC;
    """
    return self._execute_read(query, database, self._apoc_error_msg)

get_properties

get_properties(database: Optional[str] = None) -> DataFrame

Use to obtain the node and relationship properties.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing information about all nodes and relationships properties.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_properties(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the node and relationship properties.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing information about all nodes
            and relationships properties.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        CALL apoc.meta.data() YIELD label,property,type,elementType
        WHERE type<>'RELATIONSHIP'
        RETURN elementType,label,property,type
        ORDER BY elementType,label,property;
    """
    return self._execute_read(query, database, self._apoc_error_msg)

get_constraints

get_constraints(database: Optional[str] = None) -> DataFrame

Use to obtain the constraints in the graph.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing information about all the constraints.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_constraints(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the constraints in the graph.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing information about all the
            constraints.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        SHOW CONSTRAINTS
    """
    return self._execute_read(query, database, self._apoc_error_msg)

get_indexes

get_indexes(database: Optional[str] = None) -> DataFrame

Use to obtain the indexes in the graph.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing information about all the indexes.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_indexes(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the indexes in the graph.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing information about all the
            indexes.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        SHOW INDEX
    """
    return self._execute_read(query, database, self._apoc_error_msg)

get_rela_source_target_freq

get_rela_source_target_freq(database: Optional[str] = None) -> DataFrame

Use to obtain the relationships source and target frequency.

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
DataFrame

Pandas DataFrame object containing the frequency of relationships cource and target frequency.

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_rela_source_target_freq(self, database: Optional[str] = None) -> DataFrame:
    """Use to obtain the relationships source and target frequency.

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        DataFrame
            Pandas DataFrame object containing the frequency of
            relationships cource and target frequency.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        CALL apoc.meta.stats() YIELD relTypes
    """
    stat_dict = self._execute_read(query, database, self._apoc_error_msg).iloc[0,0]
    stats_dicts = []
    for key, value in stat_dict.items():
        nodes = _NODE_LABEL_RE.findall(key)
        rel_matches = _REL_TYPE_RE.findall(key)
        if len(nodes) < 2 or not rel_matches:
            continue
        info = {'sourceLabel': nodes[0], 'relationshipType': rel_matches[0],
                'targetLabel': nodes[1], 'frequency': value}
        stats_dicts.append(info)
    stats_df = DataFrame(stats_dicts)
    stats_df.sort_values(by=['relationshipType','sourceLabel','targetLabel'],inplace=True)
    stats_df.reset_index(drop=True, inplace=True)
    return stats_df

get_schema_visualization

get_schema_visualization(database: Optional[str] = None) -> Network

Use to visualize the graph data model (schema).

Parameters:

Name Type Description Default
database str

Name of the Neo4j database of which to execute the transaction. If not provided the default database is going to be use.

None

Returns:

Type Description
Network

Pyvis Network object. Render with network.write_html('schema.html').

Raises:

Type Description
ClientError

When APOC Library is not install correctly in your Neo4j deployment.

Source code in pyneoinstance/database/neo4jdbms.py
def get_schema_visualization(self, database: Optional[str] = None) -> Network:
    """Use to visualize the graph data model (schema).

        Parameters
        ----------
        database : str, optional
            Name of the Neo4j database of which to execute the transaction.
            If not provided the default database is going to be use.

        Returns
        -------
        Network
            Pyvis Network object. Render with ``network.write_html('schema.html')``.

        Raises
        ------
        ClientError
            When APOC Library is not install correctly in your Neo4j
            deployment.
    """

    query = """
        CALL apoc.meta.schema() YIELD value
    """
    schema = self._execute_read(query, database, self._apoc_error_msg).iloc[0,0]
    network = Network(cdn_resources="remote", directed=True,
                      filter_menu=True, height="800px", width="100%")
    options = """
    const options = {
        "physics": {
            "barnesHut": {
              "gravitationalConstant": -13950,
              "centralGravity": 6.15,
              "springLength": 160,
              "damping": 0.23
            },
            "minVelocity": 0.75
          }
        }
    """
    network.set_options(options)
    nodes = []
    relationships = set()
    for key in schema.keys():
        if schema[key]['type']=='node':
            nodes.append(key)
    network.add_nodes(nodes)
    for node in nodes:
        for rel in schema[node]['relationships'].keys():
            title = rel
            if schema[node]['relationships'][rel]['direction']=='in':
                target = node
                for label in schema[node]['relationships'][rel]['labels']:
                    source = label
            else:
                source = node
                for label in schema[node]['relationships'][rel]['labels']:
                    target = label
            relationships.add(source + "|" + title + "|" + target)
    for relationship in relationships:
        rela_parts = relationship.split("|")
        network.add_edge(rela_parts[0],rela_parts[2],title=rela_parts[1])
    return network

load_yaml_file

Helper to load YAML configuration files with optional key validation.

load_yaml_file

load_yaml_file(yaml_file: str, required_keys: Optional[List[str]] = None) -> Dict[str, Any]

Load the configuration file.

Parse and return the YAML file if the file is readable and formatted propertly. Otherwise it raised exceptions.

Parameters:

Name Type Description Default
yaml_file str

Full path of the YAML file.

required
required_keys List[str]

List of the required keys.

None

Returns:

Type Description
Dict[str, Any]

Python dictionary containing the YAML file data.

Raises:

Type Description
ValueError

If the file is missing the required keys.

FileNotFoundException

If the YAML file does not exists.

ParserError

If the YAML file is not formated correctly.

Source code in pyneoinstance/fileload/loadyaml.py
def load_yaml_file(yaml_file: str,
                   required_keys: Optional[List[str]] = None
                  ) -> Dict[str, Any]:
    """Load the configuration file.

    Parse and return the YAML file if the file is readable and
    formatted propertly. Otherwise it raised exceptions.

    Parameters
    ----------
    yaml_file: str
        Full path of the YAML file.
    required_keys : List[str]
        List of the required keys.

    Returns
    -------
    Dict[str, Any]
        Python dictionary containing the YAML file data.

    Raises
    ------
    ValueError
        If the file is missing the required keys.
    FileNotFoundException
        If the YAML file does not exists.
    ParserError
        If the YAML file is not formated correctly.
    """
    error_messages = {
        'FileNotFoundError': 'YAML file not found: ',
        'ParserError': 'Wrong YAML file format: ',
        'ValueError': 'Missing the following configuration key(s): '
    }
    configuration_object = None
    try:
        with open(yaml_file, encoding='utf8') as config_file:
            configuration_object = lower_dict_keys(
                yaml.load(config_file, yaml.SafeLoader))
        config_keys = configuration_object.keys()
        if required_keys:
            missing_keys = check_required_keys(required_keys, config_keys)
            if missing_keys:
                error_msg = error_messages[
                    'ValueError'] + ','.join( missing_keys)
                raise ValueError(error_msg)
    except FileNotFoundError as exception:
        error_msg = error_messages[exception.__class__.__name__] + str(exception)
        raise FileNotFoundError(error_msg) from exception
    except ParserError as exception:
        error_msg = error_messages[exception.__class__.__name__] + str(exception)
        raise ParserError(error_msg) from exception
    return configuration_object