Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
16 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions common/llm_services/base_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -840,6 +840,7 @@ def generate_cypher_prompt(self):
- Prefer attributes over primary IDs when an attribute name is more similar to the keyword in the question.
- Keep the query minimal — fewest vertex types, edge types, and attributes possible.
- Do NOT return attributes that aren't explicitly mentioned in the question. If only a vertex is mentioned, return only the vertex.
- For Jira business-key lookups where only the numeric portion is given (e.g. `2192`), use `ENDS WITH "-2192"` on the issue_key attribute rather than an exact match.
- Always include the entity from the `WHERE` clause in the final `RETURN`. Use vertex name over ID when available.
- Always use **undirected** edge patterns. Ensure edges connect correct vertex types per schema.
- Use **double quotes** for strings.
Expand Down
24 changes: 24 additions & 0 deletions docs/tutorials/configs/data_sources.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"sources": [
{
"id": "my-jira",
"type": "jira_cloud",
"enabled": true,
"display_name": "My Jira Project",
"connection": {
"site_url": "https://your-company.atlassian.net",
"email": "your-email@company.com",
"api_token": "YOUR_JIRA_API_TOKEN"
},
"scope": {
"project_keys": ["GML", "TSE"],
"created_after": null,
"updated_after": null,
"status_categories": ["new", "indeterminate", "done"],
"jql_extra": "",
"include_comments": true,
"story_points_field": null
}
}
]
}
8 changes: 7 additions & 1 deletion ecc/app/ecc_util.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,13 @@ def get_chunker(chunker_type: str = "", graphname: str = None):
chunk_size=chunker_config.get("chunk_size", 0),
overlap_size=chunker_config.get("overlap_size", -1),
)
elif chunker_type in ("structured", "markdown", "html"):
elif chunker_type in (
"structured",
"markdown",
"html",
"jira",
"jira_comment",
):
# Structure-aware chunker for markdown AND HTML: tables/figures/lists/
# code stay atomic (never split mid-row), prose char-splits by size.
# Supersedes MarkdownChunker/HTMLChunker, which split structure blindly.
Expand Down
82 changes: 77 additions & 5 deletions ecc/app/graphrag/workers.py
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,22 @@ async def chunk_doc(

v_id = doc["v_id"].lower()

# Look up the authoritative JiraIssue vertex ID from the Document's
# CONTAINS_ENTITY edge — derived IDs are unreliable for key-based vertices.
jira_issue_vertex_id: str | None = None
if chunker_type == "jira" and ":issue-doc:" in v_id:
try:
edges = await conn.getEdges(
"Document", v_id, "CONTAINS_ENTITY", "JiraIssue"
)
if edges:
jira_issue_vertex_id = edges[0]["to_id"]
except Exception as exc:
logger.warning(
f"Could not look up JiraIssue vertex for {v_id}: {exc}; "
"falling back to derived ID"
)

# Use get_chunker for all types (including images)
# For images, get_chunker returns SingleChunker which preserves markdown image references
chunker = ecc_util.get_chunker(chunker_type, graphname=conn.graphname)
Expand All @@ -173,19 +189,25 @@ async def chunk_doc(

# send chunks to be upserted (func, args)
logger.debug("chunk writes to upsert_chan")
await upsert_chan.put((upsert_chunk, (conn, v_id, chunk_id, chunk, i)))
await upsert_chan.put(
(upsert_chunk, (conn, v_id, chunk_id, chunk, i, chunker_type, jira_issue_vertex_id))
)

# send chunks to have entities extracted
logger.debug("chunk writes to extract_chan")
await extract_chan.put((chunk, chunk_id))
skip_extraction = chunker_type in ("jira", "jira_comment")
if not skip_extraction:
logger.debug("chunk writes to extract_chan")
await extract_chan.put((chunk, chunk_id))

# When extraction is enabled the extract worker pushes the
# summary-augmented embed message itself (Contextual Retrieval),
# so only embed the raw chunk here when extraction is off.
from common.config import entity_extraction_switch
if not entity_extraction_switch:
if not entity_extraction_switch or skip_extraction:
logger.debug("chunk writes to embed_chan (no extraction)")
await embed_chan.put((chunk_id, chunk, "DocumentChunk"))
if tracker is not None:
tracker.chunk_done(chunk_id)

return v_id

Expand All @@ -208,7 +230,15 @@ async def upsert_doc(conn: AsyncTigerGraphConnection, doc_id, ctype, content_tex
conn, "Document", doc_id, "HAS_CONTENT", "Content", doc_id
)

async def upsert_chunk(conn: AsyncTigerGraphConnection, doc_id, chunk_id, chunk, idx):
async def upsert_chunk(
conn: AsyncTigerGraphConnection,
doc_id,
chunk_id,
chunk,
idx,
source_type="",
jira_issue_vertex_id: "str | None" = None,
):
logger.debug(f"Upserting chunk {chunk_id}")
date_added = int(time.time())
# Build the chunk's full vertex + edge bundle and enqueue atomically.
Expand All @@ -233,6 +263,39 @@ async def upsert_chunk(conn: AsyncTigerGraphConnection, doc_id, chunk_id, chunk,
"DocumentChunk", chunk_id, "IS_AFTER",
"DocumentChunk", util.process_id(f"{doc_id}_chunk_{idx - 1}"), None,
))
if source_type == "jira" and ":issue-doc:" in doc_id:
# Link chunk to the authoritative JiraIssue vertex for graph traversal.
issue_id = jira_issue_vertex_id or doc_id.replace(":issue-doc:", ":issue:", 1)
edges.append((
"DocumentChunk",
chunk_id,
"CONTAINS_ENTITY",
"JiraIssue",
issue_id,
None,
))
elif source_type == "jira_comment" and ":comment-doc:" in doc_id:
issue_id, comment_id = doc_id.rsplit(":comment-doc:", 1)
cloud_prefix = issue_id.split(":issue:", 1)[0]
comment_vertex_id = f"{cloud_prefix}:comment:{comment_id}"
edges.extend([
(
"DocumentChunk",
chunk_id,
"CONTAINS_ENTITY",
"JiraIssue",
issue_id,
None,
),
(
"DocumentChunk",
chunk_id,
"CONTAINS_ENTITY",
"JiraComment",
comment_vertex_id,
None,
),
])
await util.upsert_group(conn, vertices, edges)


Expand Down Expand Up @@ -263,6 +326,14 @@ async def embed(
async with embed_sem:
logger.debug(f"Embedding {v_id}")

# Skip empty chunks — embedding API returns 500 on empty content.
if not content or not content.strip():
logger.warning(
f"Skipping embed for {v_id}: content is empty. "
"Check the source document for missing text."
)
return

# if loader is running, wait until it's done
if not util.loading_event.is_set():
logger.debug("Embed worker waiting for loading event to finish")
Expand All @@ -271,6 +342,7 @@ async def embed(
await embed_store.aadd_embeddings([(content, [])], [{"vertex_id": v_id}])
except Exception as e:
logger.error(f"Failed to add embeddings for {v_id}: {e}")
raise


def _is_near_duplicate(new_desc, existing_descs, threshold=0.85):
Expand Down
72 changes: 72 additions & 0 deletions ecc/tests/test_jira_comment_chunks.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,72 @@
from __future__ import annotations

import pytest

from common.embeddings.tigergraph_embedding_store import (
TigerGraphEmbeddingStore,
)
from graphrag import workers


@pytest.mark.asyncio
async def test_jira_comment_chunk_links_comment_and_issue(monkeypatch):
captured: dict = {}

async def capture_group(conn, vertices, edges):
captured["vertices"] = vertices
captured["edges"] = edges

monkeypatch.setattr(workers.util, "upsert_group", capture_group)

await workers.upsert_chunk(
object(),
"jira:cloud-1:issue:10422:comment-doc:9001",
"chunk-1",
"Short Jira comment",
0,
"jira_comment",
)

assert (
"DocumentChunk",
"chunk-1",
"CONTAINS_ENTITY",
"JiraIssue",
"jira:cloud-1:issue:10422",
None,
) in captured["edges"]
assert (
"DocumentChunk",
"chunk-1",
"CONTAINS_ENTITY",
"JiraComment",
"jira:cloud-1:comment:9001",
None,
) in captured["edges"]


def test_exhausted_embedding_retries_raise():
provider_error = RuntimeError("500 INTERNAL")

with pytest.raises(RuntimeError, match="Failed to embed chunk-1"):
TigerGraphEmbeddingStore._log_embed_failure(
"chunk-1",
provider_error,
)


@pytest.mark.asyncio
async def test_embedding_worker_propagates_store_failure():
class FailingStore:
async def aadd_embeddings(self, *args, **kwargs):
raise RuntimeError("embedding provider unavailable")

workers.util.loading_event.set()

with pytest.raises(RuntimeError, match="embedding provider unavailable"):
await workers.embed(
object(),
FailingStore(),
("chunk-1", "DocumentChunk"),
"chunk content",
)
5 changes: 5 additions & 0 deletions graphrag-ui/src/main.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ import LLMConfig from "./pages/setup/LLMConfig.tsx";
import GraphDBConfig from "./pages/setup/GraphDBConfig.tsx";
import GraphRAGConfig from "./pages/setup/GraphRAGConfig.tsx";
import McpServersConfig from "./pages/setup/McpServersConfig.tsx";
import DataSourcesConfig from "./pages/setup/DataSourcesConfig.tsx";
import CustomizePrompts from "./pages/setup/CustomizePrompts.tsx";
import { ThemeProvider } from "./components/ThemeProvider.tsx";
import { ModeToggle } from "@/components/ModeToggle.tsx";
Expand Down Expand Up @@ -79,6 +80,10 @@ const router = createBrowserRouter([
path: "kg-admin/ingest",
element: <IngestGraph />,
},
{
path: "kg-admin/data-sources",
element: <DataSourcesConfig />,
},
{
path: "server-config",
element: <Navigate to="/setup/server-config/llm" replace />,
Expand Down
Loading
Loading