Skip to content

Commit

Permalink
change chunk.status to chunk.available (#2646)
Browse files Browse the repository at this point in the history
### What problem does this PR solve?

#1102

### Type of change

- [x] New Feature (non-breaking change which adds functionality)
  • Loading branch information
JobSmithManipulation authored Sep 29, 2024
1 parent e82e8fd commit c103dd2
Show file tree
Hide file tree
Showing 3 changed files with 21 additions and 13 deletions.
4 changes: 2 additions & 2 deletions api/apps/sdk/doc.py
Original file line number Diff line number Diff line change
Expand Up @@ -609,8 +609,8 @@ def set(tenant_id):
d["content_sm_ltks"] = rag_tokenizer.fine_grained_tokenize(d["content_ltks"])
d["important_kwd"] = req["important_keywords"]
d["important_tks"] = rag_tokenizer.tokenize(" ".join(req["important_keywords"]))
if "available_int" in req:
d["available_int"] = req["available_int"]
if "available" in req:
d["available_int"] = req["available"]

try:
tenant_id = DocumentService.get_tenant_id(req["document_id"])
Expand Down
4 changes: 2 additions & 2 deletions sdk/python/ragflow/modules/chunk.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ def __init__(self, rag, res_dict):
self.knowledgebase_id = None
self.document_name = ""
self.document_id = ""
self.status = "1"
self.available = 1
for k in list(res_dict.keys()):
if k not in self.__dict__:
res_dict.pop(k)
Expand Down Expand Up @@ -39,7 +39,7 @@ def save(self) -> bool:
"content": self.content,
"important_keywords": self.important_keywords,
"document_id": self.document_id,
"status": self.status,
"available": self.available,
})
res = res.json()
if res.get("retmsg") == "success":
Expand Down
26 changes: 17 additions & 9 deletions sdk/python/test/t_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -151,14 +151,12 @@ def test_parse_and_cancel_document(self):
name3 = 'westworld.pdf'
path = 'test_data/westworld.pdf'


# Create a document in the dataset using the file path
rag.create_document(ds, name=name3, blob=open(path, "rb").read())

# Retrieve the document by name
doc = rag.get_document(name="westworld.pdf")


# Initiate asynchronous parsing
doc.async_parse()

Expand Down Expand Up @@ -231,7 +229,7 @@ def test_bulk_parse_and_cancel_documents(self):
def test_parse_document_and_chunk_list(self):
rag = RAGFlow(API_KEY, HOST_ADDRESS)
ds = rag.create_dataset(name="God7")
name='story.txt'
name = 'story.txt'
path = 'test_data/story.txt'
# name = "Test Document rag.txt"
# blob = " Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps.Sample document content for rag test66. rag wonderful apple os documents apps.Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps. Sample document content for rag test66. rag wonderful apple os documents apps."
Expand Down Expand Up @@ -266,21 +264,31 @@ def test_delete_chunk_of_chunk_list(self):
assert chunk is not None, "Chunk is None"
assert isinstance(chunk, Chunk), "Chunk was not added to chunk list"
doc = rag.get_document(name='story.txt')
chunk_count_before=doc.chunk_count
chunk_count_before = doc.chunk_count
chunk.delete()
doc = rag.get_document(name='story.txt')
assert doc.chunk_count == chunk_count_before-1, "Chunk was not deleted"
assert doc.chunk_count == chunk_count_before - 1, "Chunk was not deleted"

def test_update_chunk_content(self):
rag = RAGFlow(API_KEY, HOST_ADDRESS)
doc = rag.get_document(name='story.txt')
chunk = doc.add_chunk(content="assssddd")
assert chunk is not None, "Chunk is None"
assert isinstance(chunk, Chunk), "Chunk was not added to chunk list"
chunk.content = "ragflow123"
res=chunk.save()
assert res is True, f"Failed to update chunk, error: {res}"

res = chunk.save()
assert res is True, f"Failed to update chunk content, error: {res}"

def test_update_chunk_available(self):
rag = RAGFlow(API_KEY, HOST_ADDRESS)
doc = rag.get_document(name='story.txt')
chunk = doc.add_chunk(content="ragflow")
assert chunk is not None, "Chunk is None"
assert isinstance(chunk, Chunk), "Chunk was not added to chunk list"
chunk.available = 0
res = chunk.save()
assert res is True, f"Failed to update chunk status, error: {res}"

def test_retrieval_chunks(self):
rag = RAGFlow(API_KEY, HOST_ADDRESS)
ds = rag.create_dataset(name="God8")
Expand Down

0 comments on commit c103dd2

Please sign in to comment.