Skip to content
Open
Show file tree
Hide file tree
Changes from 10 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
104 changes: 104 additions & 0 deletions graphify/detect.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,9 @@ class FileType(str, Enum):
PAPER_EXTENSIONS = {'.pdf'}
IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'}
OFFICE_EXTENSIONS = {'.docx', '.xlsx'}
# Notebooks are converted to markdown sidecars before indexing — do NOT add .ipynb
# to CODE_EXTENSIONS or DOC_EXTENSIONS.
NOTEBOOK_EXTENSIONS = {'.ipynb'}
VIDEO_EXTENSIONS = {'.mp4', '.mov', '.webm', '.mkv', '.avi', '.m4v', '.mp3', '.wav', '.m4a', '.ogg'}

CORPUS_WARN_THRESHOLD = 50_000 # words - below this, warn "you may not need a graph"
Expand Down Expand Up @@ -517,6 +520,8 @@ def classify_file(path: Path) -> FileType | None:
return FileType.DOCUMENT
if ext in OFFICE_EXTENSIONS:
return FileType.DOCUMENT
if ext in NOTEBOOK_EXTENSIONS:
return FileType.DOCUMENT
if ext in GOOGLE_WORKSPACE_EXTENSIONS:
return FileType.DOCUMENT
if ext in VIDEO_EXTENSIONS:
Expand Down Expand Up @@ -770,6 +775,94 @@ def convert_office_file(path: Path, out_dir: Path, root: "Path | None" = None) -
return out_path


def _notebook_sidecar_path(path: Path, out_dir: Path, root: "Path | None" = None) -> Path:
"""Stable sidecar path for a converted notebook.

Uses the same scheme convert_office_file() applies to Office sources: hash
the scan-root-RELATIVE, NFC-normalized path. An absolute key would salt the
name with the checkout location, so one tracked notebook in two clones emits
two byte-identical sidecars when graphify-out/ is committed (#2059); NFC
guards macOS NFD path drift (#1226). Sources outside the scan root keep the
absolute form.
"""
import hashlib
import unicodedata
if root is None:
# Default layout: out_dir is <root>/<graphify-out>/converted.
root = out_dir.parent.parent
try:
key = path.resolve().relative_to(Path(root).resolve()).as_posix()
except (ValueError, OSError):
key = str(path.resolve())
name_hash = hashlib.sha256(unicodedata.normalize("NFC", key).encode()).hexdigest()[:8]
return out_dir / f"{path.stem}_{name_hash}.md"


def ipynb_to_markdown(path: Path) -> str:
"""Convert a Jupyter notebook to markdown, stripping outputs.

Uses the notebook's kernel language from metadata for fenced code blocks,
falling back to ``code`` when the metadata is absent.
"""
if not _file_within_size_cap(path):
return ""
try:
nb = json.loads(path.read_text(encoding="utf-8", errors="ignore"))
# Resolve the kernel language from notebook metadata so fenced code
# blocks use the correct language identifier (e.g. ```python) rather
# than the generic ```code fallback.
meta = nb.get("metadata", {})
lang = (
meta.get("language_info", {}).get("name")
or meta.get("kernelspec", {}).get("language")
or "code"
)
lines = []
for cell in nb.get("cells", []):
ct = cell.get("cell_type")
raw_src = cell.get("source", [])
src = raw_src if isinstance(raw_src, str) else "".join(raw_src)
if not src.strip():
continue
if ct == "markdown":
lines.append(src)
elif ct == "code":
lines.append(f"```{lang}\n{src}\n```")
return "\n\n".join(lines)
except Exception:
return ""


def convert_notebook_file(path: Path, out_dir: Path, root: "Path | None" = None) -> Path | None:
Comment thread
KunojiLym marked this conversation as resolved.
"""Convert a .ipynb to a markdown sidecar in out_dir.

Naming matches the Office sidecars (see _notebook_sidecar_path). The
rewrite check does not: re-running a notebook rewrites the .ipynb with
fresh outputs/execution counts while cell sources stay the same, so the
Office mtime gate would churn the sidecar. Comparing extracted markdown
keeps its mtime untouched through a re-run, and detect_incremental then
leaves an unchanged notebook alone.
"""
if path.suffix.lower() not in NOTEBOOK_EXTENSIONS:
return None

text = ipynb_to_markdown(path)
if not text.strip():
return None

out_dir.mkdir(parents=True, exist_ok=True)
out_path = _notebook_sidecar_path(path, out_dir, root=root)
payload = f"<!-- converted from {path.name} -->\n\n{text}"
try:
with open(_os_path(out_path), encoding="utf-8") as f:
if f.read() == payload:
return out_path
except OSError:
pass
out_path.write_text(payload, encoding="utf-8")
return out_path


def count_words(path: Path) -> int:
Comment thread
KunojiLym marked this conversation as resolved.
try:
ext = path.suffix.lower()
Expand Down Expand Up @@ -1385,6 +1478,17 @@ def _on_walk_error(err: OSError) -> None:
# Conversion failed (library not installed) - skip with note
skipped_sensitive.append(str(p) + " [office conversion failed - pip install graphifyy[office]]")
continue
# Notebooks: same sidecar treatment as Office files
if p.suffix.lower() in NOTEBOOK_EXTENSIONS:
md_path = convert_notebook_file(p, converted_dir, root=root)
if md_path:
if _is_ignored(md_path, root, ignore_patterns, _cache=ignore_cache):
continue
files[ftype].append(str(md_path))
total_words += _wc(md_path)
else:
skipped_sensitive.append(str(p) + " [notebook conversion failed]")
continue
files[ftype].append(str(p))
if ftype != FileType.VIDEO:
total_words += _wc(p)
Expand Down
253 changes: 253 additions & 0 deletions tests/test_detect.py
Original file line number Diff line number Diff line change
Expand Up @@ -2086,6 +2086,259 @@ def test_convert_office_file_does_not_rewrite_existing_sidecar(tmp_path, monkeyp
assert second.stat().st_mtime_ns == mtime_before


def _minimal_ipynb(cells, metadata=None):
import json
nb = {"cells": cells, "nbformat": 4, "nbformat_minor": 5}
if metadata is not None:
nb["metadata"] = metadata
return json.dumps(nb)


def test_classify_ipynb():
assert classify_file(Path("analysis.ipynb")) == FileType.DOCUMENT


def test_ipynb_to_markdown_mixed_cells(tmp_path):
nb_path = tmp_path / "nb.ipynb"
nb_path.write_text(
_minimal_ipynb(
[
{"cell_type": "markdown", "source": "# Title\n\nIntro text."},
{
"cell_type": "code",
"source": "import pandas as pd\nprint('hi')",
"outputs": [{"output_type": "stream", "text": "hi\n"}],
},
{"cell_type": "markdown", "source": "## Section"},
],
metadata={"language_info": {"name": "python"}},
),
encoding="utf-8",
)
md = detect_mod.ipynb_to_markdown(nb_path)
assert "# Title" in md
assert "```python\nimport pandas as pd" in md
assert "hi\n" not in md # outputs stripped
assert "## Section" in md
assert md.index("# Title") < md.index("```python") < md.index("## Section")


def test_ipynb_to_markdown_uses_non_python_kernel_language(tmp_path):
"""Notebooks are not Python-only: the fence must name the kernel language."""
nb_path = tmp_path / "survey.ipynb"
nb_path.write_text(
_minimal_ipynb(
[{"cell_type": "code", "source": "summary(df)"}],
metadata={"kernelspec": {"language": "R"}},
),
encoding="utf-8",
)
assert "```R\nsummary(df)" in detect_mod.ipynb_to_markdown(nb_path)


def test_ipynb_to_markdown_falls_back_to_generic_fence(tmp_path):
"""A notebook with no language metadata still fences its code cells."""
nb_path = tmp_path / "bare.ipynb"
nb_path.write_text(
_minimal_ipynb([{"cell_type": "code", "source": "x = 1"}]),
encoding="utf-8",
)
assert "```code\nx = 1" in detect_mod.ipynb_to_markdown(nb_path)


def test_ipynb_to_markdown_accepts_string_source(tmp_path):
"""nbformat allows `source` as a plain string as well as a list of lines."""
import json

nb_path = tmp_path / "str.ipynb"
nb_path.write_text(
json.dumps({
"cells": [{"cell_type": "code", "source": "x = 1"}],
"metadata": {"language_info": {"name": "python"}},
"nbformat": 4,
"nbformat_minor": 5,
}),
encoding="utf-8",
)
assert "```python\nx = 1" in detect_mod.ipynb_to_markdown(nb_path)


def test_ipynb_to_markdown_empty_notebook(tmp_path):
nb_path = tmp_path / "empty.ipynb"
nb_path.write_text(_minimal_ipynb([]), encoding="utf-8")
assert detect_mod.ipynb_to_markdown(nb_path) == ""


def test_ipynb_to_markdown_malformed_json(tmp_path):
nb_path = tmp_path / "bad.ipynb"
nb_path.write_text("{not valid json", encoding="utf-8")
assert detect_mod.ipynb_to_markdown(nb_path) == ""


def test_detect_converts_notebook_to_sidecar(tmp_path):
nb_path = tmp_path / "analysis.ipynb"
nb_path.write_text(
_minimal_ipynb([
{"cell_type": "markdown", "source": "# Analysis"},
{"cell_type": "code", "source": "x = 1"},
]),
encoding="utf-8",
)
result = detect(tmp_path)
assert len(result["files"]["document"]) == 1
sidecar = Path(result["files"]["document"][0])
assert sidecar.suffix == ".md"
assert sidecar.exists()
text = sidecar.read_text(encoding="utf-8")
assert "converted from analysis.ipynb" in text
assert "# Analysis" in text
assert "x = 1" in text
assert result["total_words"] > 0


def test_convert_notebook_file_rewrite_semantics(tmp_path):
"""Single entry-point for convert_notebook_file unit coverage.

Keeps afferent coupling on the converter low (one test caller + detect)
while still checking empty notebooks, output-only re-runs, and source edits.
"""
out_dir = tmp_path / "converted"

empty = tmp_path / "empty.ipynb"
empty.write_text(_minimal_ipynb([]), encoding="utf-8")
assert detect_mod.convert_notebook_file(empty, out_dir) is None
assert not list(out_dir.glob("*.md"))

nb_path = tmp_path / "analysis.ipynb"
nb_path.write_text(
_minimal_ipynb([{"cell_type": "code", "source": "print(1)", "outputs": []}]),
encoding="utf-8",
)
sidecar = detect_mod.convert_notebook_file(nb_path, out_dir)
assert sidecar is not None
mtime_before = sidecar.stat().st_mtime_ns

nb_path.write_text(
_minimal_ipynb([
{
"cell_type": "code",
"source": "print(1)",
"outputs": [{"output_type": "stream", "text": "1\n"}],
"execution_count": 1,
},
]),
encoding="utf-8",
)
again = detect_mod.convert_notebook_file(nb_path, out_dir)
assert again == sidecar
assert again.stat().st_mtime_ns == mtime_before

nb_path.write_text(
_minimal_ipynb([{"cell_type": "code", "source": "x = 2", "outputs": []}]),
encoding="utf-8",
)
updated = detect_mod.convert_notebook_file(nb_path, out_dir)
assert updated == sidecar
assert "x = 2" in updated.read_text(encoding="utf-8")
assert updated.stat().st_mtime_ns >= mtime_before


def test_detect_refreshes_notebook_sidecar_on_source_change(tmp_path):
"""Cell source edits must update the sidecar so a later extract sees new content."""
nb_path = tmp_path / "analysis.ipynb"
nb_path.write_text(
_minimal_ipynb([{"cell_type": "markdown", "source": "v1"}]),
encoding="utf-8",
)
detect(tmp_path)
converted_dir = tmp_path / "graphify-out" / "converted"
sidecar = next(converted_dir.glob("analysis_*.md"))

nb_path.write_text(
_minimal_ipynb([{"cell_type": "markdown", "source": "v2 updated"}]),
encoding="utf-8",
)
detect(tmp_path)
assert "v2 updated" in sidecar.read_text(encoding="utf-8")


def test_detect_incremental_ignores_notebook_output_only_changes(tmp_path):
import json

nb_path = tmp_path / "analysis.ipynb"
nb_path.write_text(
_minimal_ipynb([{"cell_type": "code", "source": "print(1)", "outputs": []}]),
encoding="utf-8",
)
first = detect(tmp_path)
sidecar = Path(first["files"]["document"][0])
mtime_before = sidecar.stat().st_mtime_ns
manifest_path = tmp_path / "graphify-out" / "manifest.json"
Path(manifest_path).write_text(
json.dumps({
str(sidecar): {
"mtime": sidecar.stat().st_mtime,
"ast_hash": "a" * 32,
"semantic_hash": "b" * 32,
}
}),
encoding="utf-8",
)

nb_path.write_text(
_minimal_ipynb([
{
"cell_type": "code",
"source": "print(1)",
"outputs": [{"output_type": "stream", "text": "1\n"}],
"execution_count": 1,
},
]),
encoding="utf-8",
)
inc = detect_incremental(tmp_path, manifest_path=str(manifest_path))
assert sidecar.stat().st_mtime_ns == mtime_before
assert not inc["new_files"]["document"]
assert str(sidecar) in inc["unchanged_files"]["document"]


def test_notebook_sidecar_path_stable_across_checkouts_and_stems(tmp_path):
"""#2059: notebook sidecar names come from scan-root-relative paths."""
def _name(root, rel):
src = root / rel
src.parent.mkdir(parents=True, exist_ok=True)
src.write_text("placeholder", encoding="utf-8")
return detect_mod._notebook_sidecar_path(
src, root / "graphify-out" / "converted", root=root
).name

assert _name(tmp_path / "checkout-a", "notebooks/analysis.ipynb") == _name(
tmp_path / "somewhere-else" / "checkout-b", "notebooks/analysis.ipynb"
)

root = tmp_path / "repo"
name_a = _name(root, "a/analysis.ipynb")
name_b = _name(root, "b/analysis.ipynb")
assert name_a != name_b

# Outside the scan root: absolute fallback is deterministic.
out_dir = root / "graphify-out" / "converted"
outside = tmp_path / "elsewhere" / "analysis.ipynb"
outside.parent.mkdir(parents=True)
outside.write_text("x", encoding="utf-8")
assert detect_mod._notebook_sidecar_path(
outside, out_dir, root=root
) == detect_mod._notebook_sidecar_path(outside, out_dir, root=root)

# No explicit root -> out_dir.parent.parent fallback matches explicit root.
checkout = tmp_path / "checkout-a"
src = checkout / "notebooks" / "analysis.ipynb"
converted = checkout / "graphify-out" / "converted"
explicit = detect_mod._notebook_sidecar_path(src, converted, root=checkout)
fallback = detect_mod._notebook_sidecar_path(src, converted)
assert explicit.name == fallback.name


def test_convert_office_file_sidecar_name_stable_across_checkouts(tmp_path, monkeypatch):
"""#2059: the sidecar name must depend on the scan-root-RELATIVE path, not the
absolute checkout location, so the same tracked file in two clones/worktrees
Expand Down