From c49568dd3494776edb2a33d5a0117707ab1850ca Mon Sep 17 00:00:00 2001 From: Cyber MacGeddon Date: Mon, 5 May 2025 09:40:02 +0100 Subject: [PATCH] Restructuring API a little --- trustgraph-base/trustgraph/api/api.py | 231 ++++++++++-------- .../scripts/tg-add-library-document | 4 +- .../scripts/tg-remove-library-document | 59 +++++ .../scripts/tg-show-library-documents | 4 +- trustgraph-cli/setup.py | 1 + 5 files changed, 191 insertions(+), 108 deletions(-) create mode 100755 trustgraph-cli/scripts/tg-remove-library-document diff --git a/trustgraph-base/trustgraph/api/api.py b/trustgraph-base/trustgraph/api/api.py index bfe774a4..b8f85eb9 100644 --- a/trustgraph-base/trustgraph/api/api.py +++ b/trustgraph-base/trustgraph/api/api.py @@ -118,92 +118,8 @@ class Api: except: raise ProtocolException(f"Response not formatted correctly") - def library_add_document( - self, document, id, metadata, user, title, comments, - kind="text/plain", tags=[], - ): - - if id is None: - - if metadata is not None: - - # Situation makes no sense. What can the metadata possibly - # mean if the caller doesn't know the document ID. - # Metadata should relate to the document by ID - raise RuntimeError("Can't specify metadata without id") - - id = hash(document) - - if not title: title = "" - if not comments: comments = "" - - triples = [] - - def emit(t): - triples.append(t) - - if metadata: - metadata.emit( - lambda t: triples.append({ - "s": { "v": t["s"], "e": isinstance(t["s"], Uri) }, - "p": { "v": t["p"], "e": isinstance(t["p"], Uri) }, - "o": { "v": t["o"], "e": isinstance(t["o"], Uri) } - }) - ) - - input = { - "operation": "add-document", - "document-metadata": { - "id": id, - "time": int(time.time()), - "kind": kind, - "title": title, - "comments": comments, - "metadata": triples, - "user": user, - "tags": tags - }, - "content": base64.b64encode(document).decode("utf-8"), - } - - return self.request( - f"librarian", - input - ) - - - def library_get_documents(self, user): - - input = { - "operation": "list-documents", - "user": user, - } - - object = self.request("librarian", input) - - try: - return [ - DocumentMetadata( - id = v["id"], - time = datetime.datetime.fromtimestamp(v["time"]), - kind = v["kind"], - title = v["title"], - comments = v.get("comments", ""), - metadata = [ - Triple( - s = to_value(w["s"]), - p = to_value(w["p"]), - o = to_value(w["o"]) - ) - for w in v["metadata"] - ], - user = v["user"], - tags = v["tags"] - ) - for v in object["document-metadatas"] - ] - except: - raise ProtocolException(f"Response not formatted correctly") + def library(self): + return Library(self) def config_get(self, keys): @@ -356,12 +272,119 @@ class Api: self.request("flow", input) +class Library: + + def __init__(self, api): + self.api = api + + def request(self, request): + return self.api.request(f"librarian", request) + + def add_document( + self, document, id, metadata, user, title, comments, + kind="text/plain", tags=[], + ): + + if id is None: + + if metadata is not None: + + # Situation makes no sense. What can the metadata possibly + # mean if the caller doesn't know the document ID. + # Metadata should relate to the document by ID + raise RuntimeError("Can't specify metadata without id") + + id = hash(document) + + if not title: title = "" + if not comments: comments = "" + + triples = [] + + def emit(t): + triples.append(t) + + if metadata: + metadata.emit( + lambda t: triples.append({ + "s": { "v": t["s"], "e": isinstance(t["s"], Uri) }, + "p": { "v": t["p"], "e": isinstance(t["p"], Uri) }, + "o": { "v": t["o"], "e": isinstance(t["o"], Uri) } + }) + ) + + input = { + "operation": "add-document", + "document-metadata": { + "id": id, + "time": int(time.time()), + "kind": kind, + "title": title, + "comments": comments, + "metadata": triples, + "user": user, + "tags": tags + }, + "content": base64.b64encode(document).decode("utf-8"), + } + + return self.request(input) + + def get_documents(self, user): + + input = { + "operation": "list-documents", + "user": user, + } + + object = self.request(input) + + try: + return [ + DocumentMetadata( + id = v["id"], + time = datetime.datetime.fromtimestamp(v["time"]), + kind = v["kind"], + title = v["title"], + comments = v.get("comments", ""), + metadata = [ + Triple( + s = to_value(w["s"]), + p = to_value(w["p"]), + o = to_value(w["o"]) + ) + for w in v["metadata"] + ], + user = v["user"], + tags = v["tags"] + ) + for v in object["document-metadatas"] + ] + except: + raise ProtocolException(f"Response not formatted correctly") + + def remove_document(self, user, id): + + input = { + "operation": "remove-document", + "user": user, + "document-id": id, + } + + object = self.request(input) + + return {} + class Flow: def __init__(self, api, flow): self.api = api self.flow = flow + def request(self, path, request): + + return self.api.request(f"flow/{self.flow}/{path}", request) + def text_completion(self, system, prompt): # The input consists of system and prompt strings @@ -370,8 +393,8 @@ class Flow: "prompt": prompt } - return self.api.request( - f"flow/{self.flow}/service/text-completion", + return self.request( + "service/text-completion", input )["response"] @@ -382,8 +405,8 @@ class Flow: "question": question } - return self.api.request( - f"flow/{self.flow}/service/agent", + return self.request( + "service/agent", input )["answer"] @@ -404,8 +427,8 @@ class Flow: "max-path-length": max_path_length, } - return self.api.request( - f"flow/{self.flow}/service/graph-rag", + return self.request( + "service/graph-rag", input )["response"] @@ -422,8 +445,8 @@ class Flow: "doc-limit": doc_limit, } - return self.api.request( - f"flow/{self.flow}/service/document-rag", + return self.request( + "service/document-rag", input )["response"] @@ -434,8 +457,8 @@ class Flow: "text": text } - return self.api.request( - f"flow/{self.flow}/service/embeddings", + return self.request( + "service/embeddings", input )["vectors"] @@ -447,8 +470,8 @@ class Flow: "variables": variables } - object = self.api.request( - f"flow/{self.flow}/service/prompt", + object = self.request( + "service/prompt", input ) @@ -496,8 +519,8 @@ class Flow: raise RuntimeError("o must be Uri or Literal") input["o"] = { "v": str(o), "e": isinstance(o, Uri), } - object = self.api.request( - f"flow/{self.flow}/service/triples", + object = self.request( + "service/triples", input ) @@ -552,8 +575,8 @@ class Flow: if collection: input["collection"] = collection - return self.api.request( - f"flow/{self.flow}/service/document-load", + return self.request( + "service/document-load", input ) @@ -597,8 +620,8 @@ class Flow: if collection: input["collection"] = collection - return self.api.request( - f"flow/{self.flow}/service/text-load", + return self.request( + "service/text-load", input ) diff --git a/trustgraph-cli/scripts/tg-add-library-document b/trustgraph-cli/scripts/tg-add-library-document index 9bb83487..a7032912 100755 --- a/trustgraph-cli/scripts/tg-add-library-document +++ b/trustgraph-cli/scripts/tg-add-library-document @@ -25,7 +25,7 @@ class Loader: self, url, user, metadata, title, comments, kind, tags ): - self.api = Api(url) + self.api = Api(url).library() self.user = user self.metadata = metadata @@ -57,7 +57,7 @@ class Loader: self.metadata.id = id - self.api.library_add_document( + self.api.add_document( document=data, id=id, metadata=self.metadata, user=self.user, kind=self.kind, title=self.title, comments=self.comments, tags=self.tags diff --git a/trustgraph-cli/scripts/tg-remove-library-document b/trustgraph-cli/scripts/tg-remove-library-document new file mode 100755 index 00000000..c714d1b6 --- /dev/null +++ b/trustgraph-cli/scripts/tg-remove-library-document @@ -0,0 +1,59 @@ +#!/usr/bin/env python3 + +""" +Remove a PDF document from the library +""" + +import argparse +import os +import uuid + +from trustgraph.api import Api + +default_url = os.getenv("TRUSTGRAPH_URL", 'http://localhost:8088/') +default_user = 'trustgraph' + + +def remove_doc(url, user, id): + + api = Api(url).library() + + api.remove_document(user=user, id=id) + +def main(): + + parser = argparse.ArgumentParser( + prog='tg-remove-library-document', + description=__doc__, + ) + + parser.add_argument( + '-u', '--url', + default=default_url, + help=f'API URL (default: {default_url})', + ) + + parser.add_argument( + '-U', '--user', + default=default_user, + help=f'User ID (default: {default_user})' + ) + + parser.add_argument( + '--identifier', '--id', + required=True, + help=f'Document ID' + ) + + args = parser.parse_args() + + try: + + remove_doc(args.url, args.user, args.identifier) + + except Exception as e: + + print("Exception:", e, flush=True) + +main() + diff --git a/trustgraph-cli/scripts/tg-show-library-documents b/trustgraph-cli/scripts/tg-show-library-documents index 63962182..d71261c4 100644 --- a/trustgraph-cli/scripts/tg-show-library-documents +++ b/trustgraph-cli/scripts/tg-show-library-documents @@ -14,9 +14,9 @@ default_user = "trustgraph" def show_docs(url, user): - api = Api(url) + api = Api(url).library() - docs = api.library_get_documents(user=user) + docs = api.get_documents(user=user) if len(docs) == 0: print("No documents.") diff --git a/trustgraph-cli/setup.py b/trustgraph-cli/setup.py index 63de9450..11a16790 100644 --- a/trustgraph-cli/setup.py +++ b/trustgraph-cli/setup.py @@ -72,6 +72,7 @@ setuptools.setup( "scripts/tg-show-flows", "scripts/tg-show-library-documents", "scripts/tg-add-library-document", + "scripts/tg-remove-library-document", "scripts/tg-show-processor-state", "scripts/tg-show-prompts", "scripts/tg-show-token-costs",