Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -209,7 +209,7 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg

| Type | Extensions |
|------|-----------|
| Code (31 languages) | `.py .ts .js .jsx .tsx .mjs .go .rs .java .c .cpp .h .hpp .rb .cs .kt .scala .php .swift .lua .luau .zig .ps1 .ex .exs .m .mm .jl .vue .svelte .astro .groovy .gradle .dart .v .sv .sql .f .f90 .f95 .f03 .f08 .pas .pp .dpr .dpk .lpr .inc .dfm .lfm .lpk .sh .bash .json` |
| Code (31 languages) | `.py .ts .js .jsx .tsx .mjs .go .rs .java .c .cpp .h .hpp .rb .cs .kt .scala .php .swift .lua .luau .zig .ps1 .ex .exs .m .mm .jl .vue .svelte .astro .groovy .gradle .dart .v .sv .sql .f .f90 .f95 .f03 .f08 .pas .pp .dpr .dpk .lpr .inc .dfm .lfm .lpk .sh .bash .json .sln .csproj .fsproj .vbproj .razor .cshtml` |
| Docs | `.md .mdx .qmd .html .txt .rst .yaml .yml` |
| Office | `.docx .xlsx` (requires `pip install graphifyy[office]`) |
| Google Workspace | `.gdoc .gsheet .gslides` (opt-in; requires `gws` auth and `--google-workspace`; Sheets need `pip install graphifyy[google]`) |
Expand Down
2 changes: 1 addition & 1 deletion graphify/detect.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ class FileType(str, Enum):

_MANIFEST_PATH = "graphify-out/manifest.json"

CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json'}
CODE_EXTENSIONS = {'.py', '.ts', '.tsx', '.js', '.jsx', '.mjs', '.ejs', '.ets', '.go', '.rs', '.java', '.groovy', '.gradle', '.cpp', '.cc', '.cxx', '.c', '.h', '.hpp', '.rb', '.swift', '.kt', '.kts', '.cs', '.scala', '.php', '.lua', '.luau', '.toc', '.zig', '.ps1', '.ex', '.exs', '.m', '.mm', '.jl', '.vue', '.svelte', '.astro', '.dart', '.v', '.sv', '.sql', '.r', '.f', '.F', '.f90', '.F90', '.f95', '.F95', '.f03', '.F03', '.f08', '.F08', '.pas', '.pp', '.dpr', '.dpk', '.lpr', '.inc', '.dfm', '.lfm', '.lpk', '.sh', '.bash', '.json', '.sln', '.csproj', '.fsproj', '.vbproj', '.razor', '.cshtml'}
DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.txt', '.rst', '.html', '.yaml', '.yml'}
PAPER_EXTENSIONS = {'.pdf'}
IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
Expand Down
297 changes: 297 additions & 0 deletions graphify/extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -7695,6 +7695,297 @@ def _prescan_functions(node) -> None:
return {"nodes": nodes, "edges": edges}


# ── .NET project files (.sln, .csproj, .razor) ──────────────────────────────

def extract_sln(path: Path) -> dict:
"""Extract projects and inter-project dependencies from a .sln file."""
try:
src = path.read_text(encoding="utf-8", errors="replace")
except OSError:
return {"nodes": [], "edges": [], "error": f"cannot read {path}"}

file_nid = _make_id(str(path))
str_path = str(path)
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = set()
seen_ids.add(file_nid)

_PROJECT_RE = re.compile(
r'Project\("[^"]*"\)\s*=\s*"([^"]+)"\s*,\s*"([^"]+)"\s*,\s*"([^"]*)"'
)
_DEP_RE = re.compile(r'\{([0-9a-fA-F-]+)\}\s*=\s*\{([0-9a-fA-F-]+)\}')

guid_to_nid: dict[str, str] = {}

for m in _PROJECT_RE.finditer(src):
proj_name = m.group(1)
proj_path = m.group(2).replace("\\", "/")
proj_guid = m.group(3).strip("{}")

proj_nid = _make_id(proj_path)
if proj_nid and proj_nid not in seen_ids:
seen_ids.add(proj_nid)
nodes.append({"id": proj_nid, "label": proj_name,
"file_type": "code", "source_file": proj_path,
"source_location": None})
edges.append({"source": file_nid, "target": proj_nid,
"relation": "contains", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})
if proj_guid:
guid_to_nid[proj_guid.lower()] = proj_nid

in_dep_section = False
current_proj_guid: str | None = None
_PROJECT_LINE_RE = re.compile(r'Project\("[^"]*"\)\s*=\s*"[^"]+"\s*,\s*"[^"]+"\s*,\s*"\{([^}]+)\}"')
for line in src.splitlines():
proj_line_m = _PROJECT_LINE_RE.search(line)
if proj_line_m:
current_proj_guid = proj_line_m.group(1).lower()
continue
if line.strip() == "EndProject":
current_proj_guid = None
continue
if "ProjectSection(ProjectDependencies)" in line:
in_dep_section = True
continue
if in_dep_section and "EndProjectSection" in line:
in_dep_section = False
continue
if in_dep_section and current_proj_guid:
dep_m = _DEP_RE.search(line)
if dep_m:
to_guid = dep_m.group(1).lower()
from_nid = guid_to_nid.get(current_proj_guid)
to_nid = guid_to_nid.get(to_guid)
if from_nid and to_nid and from_nid != to_nid:
edges.append({"source": from_nid, "target": to_nid,
"relation": "imports", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

return {"nodes": nodes, "edges": edges}


def extract_csproj(path: Path) -> dict:
"""Extract packages, project refs, and target framework from a .csproj/.fsproj/.vbproj."""
import xml.etree.ElementTree as ET

try:
src = path.read_bytes()
except OSError:
return {"nodes": [], "edges": [], "error": f"cannot read {path}"}

if len(src) > 2_097_152:
return {"nodes": [], "edges": [], "error": "project file too large"}

try:
tree = ET.fromstring(src)
except ET.ParseError as e:
return {"nodes": [], "edges": [], "error": f"XML parse error: {e}"}

file_nid = _make_id(str(path))
str_path = str(path)
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = set()
seen_ids.add(file_nid)

ns = ""
root_tag = tree.tag
if root_tag.startswith("{"):
ns = root_tag.split("}")[0] + "}"

def find_all(tag: str):
return tree.iter(f"{ns}{tag}")

for tf in find_all("TargetFramework"):
if tf.text:
fw_nid = _make_id("framework", tf.text.strip())
if fw_nid and fw_nid not in seen_ids:
seen_ids.add(fw_nid)
nodes.append({"id": fw_nid, "label": tf.text.strip(),
"file_type": "concept", "source_file": str_path,
"source_location": None})
edges.append({"source": file_nid, "target": fw_nid,
"relation": "references", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

for tf in find_all("TargetFrameworks"):
if tf.text:
for fw in tf.text.strip().split(";"):
fw = fw.strip()
if fw:
fw_nid = _make_id("framework", fw)
if fw_nid and fw_nid not in seen_ids:
seen_ids.add(fw_nid)
nodes.append({"id": fw_nid, "label": fw,
"file_type": "concept", "source_file": str_path,
"source_location": None})
edges.append({"source": file_nid, "target": fw_nid,
"relation": "references", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

for pkg in find_all("PackageReference"):
name = pkg.get("Include") or pkg.get("include") or ""
version = pkg.get("Version") or pkg.get("version") or ""
if not name:
continue
pkg_nid = _make_id("nuget", name)
label = f"{name} ({version})" if version else name
if pkg_nid and pkg_nid not in seen_ids:
seen_ids.add(pkg_nid)
nodes.append({"id": pkg_nid, "label": label,
"file_type": "code", "source_file": str_path,
"source_location": None})
edges.append({"source": file_nid, "target": pkg_nid,
"relation": "imports", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

for proj in find_all("ProjectReference"):
ref_path = proj.get("Include") or proj.get("include") or ""
if not ref_path:
continue
ref_path_norm = ref_path.replace("\\", "/")
proj_nid = _make_id(ref_path_norm)
if proj_nid and proj_nid not in seen_ids:
seen_ids.add(proj_nid)
proj_label = Path(ref_path_norm).name
nodes.append({"id": proj_nid, "label": proj_label,
"file_type": "code", "source_file": ref_path_norm,
"source_location": None})
edges.append({"source": file_nid, "target": proj_nid,
"relation": "imports", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

sdk = tree.get("Sdk") or ""
if sdk:
sdk_nid = _make_id("sdk", sdk)
if sdk_nid and sdk_nid not in seen_ids:
seen_ids.add(sdk_nid)
nodes.append({"id": sdk_nid, "label": sdk,
"file_type": "concept", "source_file": str_path,
"source_location": None})
edges.append({"source": file_nid, "target": sdk_nid,
"relation": "references", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

return {"nodes": nodes, "edges": edges}


def extract_razor(path: Path) -> dict:
"""Extract directives, component refs, and @code methods from .razor/.cshtml."""
try:
src = path.read_text(encoding="utf-8", errors="replace")
except OSError:
return {"nodes": [], "edges": [], "error": f"cannot read {path}"}

file_nid = _make_id(str(path))
str_path = str(path)
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
"source_file": str_path, "source_location": None}]
edges: list[dict] = []
seen_ids: set[str] = set()
seen_ids.add(file_nid)

def _add_ref(target_name: str, relation: str, line: int) -> None:
tgt_nid = _make_id(target_name)
if not tgt_nid:
return
if tgt_nid not in seen_ids:
seen_ids.add(tgt_nid)
nodes.append({"id": tgt_nid, "label": target_name,
"file_type": "code", "source_file": str_path,
"source_location": f"L{line}"})
edges.append({"source": file_nid, "target": tgt_nid,
"relation": relation, "confidence": "EXTRACTED",
"source_file": str_path, "source_location": f"L{line}",
"weight": 1.0})

for i, line in enumerate(src.splitlines(), 1):
m = re.match(r'@using\s+([\w.]+)', line)
if m:
_add_ref(m.group(1), "imports", i)
continue

m = re.match(r'@inject\s+([\w.<>\[\]]+)\s+(\w+)', line)
if m:
_add_ref(m.group(1), "imports", i)
continue

m = re.match(r'@inherits\s+([\w.<>\[\]]+)', line)
if m:
_add_ref(m.group(1), "inherits", i)
continue

m = re.match(r'@model\s+([\w.<>\[\]]+)', line)
if m:
_add_ref(m.group(1), "references", i)
continue

m = re.match(r'@page\s+"([^"]+)"', line)
if m:
route = m.group(1)
route_nid = _make_id("route", route)
if route_nid and route_nid not in seen_ids:
seen_ids.add(route_nid)
nodes.append({"id": route_nid, "label": f"route:{route}",
"file_type": "concept", "source_file": str_path,
"source_location": f"L{i}"})
edges.append({"source": file_nid, "target": route_nid,
"relation": "references", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})
continue

_COMPONENT_RE = re.compile(r'<([A-Z][A-Za-z0-9]+)[\s/>]')
_HTML_TAGS = frozenset({
"DOCTYPE", "Html", "Head", "Body", "Div", "Span", "Table", "Form",
"Input", "Button", "Select", "Option", "Label", "Textarea",
"Script", "Style", "Link", "Meta", "Title", "Header", "Footer",
"Nav", "Main", "Section", "Article", "Aside",
})
for m in _COMPONENT_RE.finditer(src):
comp_name = m.group(1)
if comp_name in _HTML_TAGS:
continue
line_num = src[:m.start()].count("\n") + 1
_add_ref(comp_name, "calls", line_num)

_CODE_BLOCK_RE = re.compile(r'@code\s*\{', re.MULTILINE)
for m in _CODE_BLOCK_RE.finditer(src):
block_start = m.end()
depth = 1
pos = block_start
while pos < len(src) and depth > 0:
if src[pos] == '{':
depth += 1
elif src[pos] == '}':
depth -= 1
pos += 1
code_block = src[block_start:pos - 1] if depth == 0 else ""

_METHOD_RE = re.compile(
r'(?:public|private|protected|internal|static|async|override|virtual|abstract)\s+'
r'[\w<>\[\],\s]+\s+(\w+)\s*\('
)
for mm in _METHOD_RE.finditer(code_block):
method_name = mm.group(1)
abs_pos = block_start + mm.start()
method_line = src[:abs_pos].count("\n") + 1
method_nid = _make_id(_file_stem(path), method_name)
if method_nid and method_nid not in seen_ids:
seen_ids.add(method_nid)
nodes.append({"id": method_nid, "label": method_name,
"file_type": "code", "source_file": str_path,
"source_location": f"L{method_line}"})
edges.append({"source": file_nid, "target": method_nid,
"relation": "contains", "confidence": "EXTRACTED",
"source_file": str_path, "weight": 1.0})

return {"nodes": nodes, "edges": edges}


def extract_json(path: Path) -> dict:
"""Extract top-level keys, nested structure, and dependency edges from a .json file."""
_JSON_MAX_BYTES = 1_048_576 # 1 MiB — skip large fixture dumps / GeoJSON blobs
Expand Down Expand Up @@ -7907,6 +8198,12 @@ def walk_object(obj_node, parent_nid: str, parent_key: str | None,
".sh": extract_bash,
".bash": extract_bash,
".json": extract_json,
".sln": extract_sln,
".csproj": extract_csproj,
".fsproj": extract_csproj,
".vbproj": extract_csproj,
".razor": extract_razor,
".cshtml": extract_razor,
}


Expand Down
21 changes: 21 additions & 0 deletions tests/fixtures/sample.csproj
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
<Project Sdk="Microsoft.NET.Sdk.Web">

<PropertyGroup>
<TargetFramework>net8.0</TargetFramework>
<Nullable>enable</Nullable>
<ImplicitUsings>enable</ImplicitUsings>
</PropertyGroup>

<ItemGroup>
<PackageReference Include="Microsoft.AspNetCore.Authentication.JwtBearer" Version="8.0.0" />
<PackageReference Include="Swashbuckle.AspNetCore" Version="6.5.0" />
<PackageReference Include="MediatR" Version="12.2.0" />
<PackageReference Include="FluentValidation" Version="11.9.0" />
</ItemGroup>

<ItemGroup>
<ProjectReference Include="..\Domain\Domain.csproj" />
<ProjectReference Include="..\Infrastructure\Infrastructure.csproj" />
</ItemGroup>

</Project>
36 changes: 36 additions & 0 deletions tests/fixtures/sample.razor
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
@page "/counter"
@using Microsoft.AspNetCore.Components
@using MyApp.Services
@inject ICounterService CounterService
@inject NavigationManager Navigation
@inherits ComponentBase

<h1>Counter</h1>

<p>Current count: @currentCount</p>

<Button OnClick="IncrementCount">Click me</Button>
<WeatherDisplay City="@city" />
<DataGrid TItem="CounterRecord" Items="@records" />

@code {
private int currentCount = 0;
private string city = "Seattle";
private List<CounterRecord> records = new();

private void IncrementCount()
{
currentCount++;
CounterService.Increment();
}

public async Task LoadData()
{
records = await CounterService.GetRecords();
}

protected override async Task OnInitializedAsync()
{
await LoadData();
}
}
Loading