Skip to content

Commit 731a86d

Browse files
committed
add a small helper to auto generate internal links from matrix.org links
Signed-off-by: MTRNord <MTRNord@users.noreply.github.com>
1 parent 62a01bb commit 731a86d

2 files changed

Lines changed: 200 additions & 0 deletions

File tree

‎.gitignore‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,3 +2,4 @@
22
.DS_Store
33
gatsby
44
.vscode
5+
__pycache__
Lines changed: 199 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,199 @@
1+
#!/usr/bin/env python3
2+
# SPDX-License-Identifier: Apache-2.0
3+
# SPDX-FileCopyrightText: 2026 The Matrix.org Foundation C.I.C.
4+
5+
"""
6+
Convert https://matrix.org links to Zola internal link format (@/).
7+
8+
This script converts absolute matrix.org links to relative Zola links using the
9+
@/ prefix, which references content from the content/ directory. It intelligently
10+
resolves file paths by checking the actual directory structure.
11+
12+
Usage:
13+
python contrib/convert_matrix_org_links.py <file_path>
14+
python contrib/convert_matrix_org_links.py content/blog/2026/05/2026-05-04-twim.md
15+
"""
16+
17+
import re
18+
import sys
19+
from pathlib import Path
20+
21+
22+
def find_file_in_content(path_segment: str, content_root: Path) -> str | None:
23+
"""
24+
Find the actual file path in content/ directory for a given path segment.
25+
26+
Args:
27+
path_segment: Path segment from URL (e.g., "blog/2026/05/conf-ticket-sales")
28+
content_root: Root of the content directory
29+
30+
Returns:
31+
Relative path from content root to the file, or None if not found
32+
"""
33+
# Try with .md extension
34+
candidate = content_root / f"{path_segment}.md"
35+
if candidate.exists():
36+
return f"@/{path_segment}.md"
37+
38+
# Try with index.md
39+
candidate = content_root / path_segment / "index.md"
40+
if candidate.exists():
41+
result = f"@/{path_segment}/index.md"
42+
return result.replace("//", "/")
43+
44+
# Try with _index.md
45+
candidate = content_root / path_segment / "_index.md"
46+
if candidate.exists():
47+
result = f"@/{path_segment}/_index.md"
48+
return result.replace("//", "/")
49+
50+
# Try to find a matching file with date prefix (for blog posts)
51+
if "blog/" in path_segment:
52+
parts = path_segment.split("/")
53+
if len(parts) >= 3: # e.g., ["blog", "2026", "05", "some-post"]
54+
dir_path = content_root / "/".join(parts[:-1])
55+
if dir_path.exists():
56+
slug = parts[-1]
57+
# Look for files matching pattern YYYY-MM-DD-{slug}.md
58+
for file in dir_path.glob(f"*-{slug}.md"):
59+
result = f"@/{'/'.join(parts[:-1])}/{file.name}"
60+
return result.replace("//", "/")
61+
62+
return None
63+
64+
65+
def extract_matrix_org_path(url: str) -> tuple[str | None, str | None]:
66+
"""
67+
Extract the path and anchor from a matrix.org URL.
68+
69+
Args:
70+
url: Full URL starting with https://matrix.org
71+
72+
Returns:
73+
Tuple of (path, anchor) or (None, None) if not a matrix.org link
74+
"""
75+
match = re.match(r"https://matrix\.org(/[^)>\s#]*)(?:#([^)>\s]*))?", url)
76+
if match:
77+
return match.group(1), match.group(2)
78+
return None, None
79+
80+
81+
def convert_matrix_org_links(file_path: str, content_root: str | None = None) -> bool:
82+
"""
83+
Convert all https://matrix.org links in a markdown file to Zola format.
84+
85+
Args:
86+
file_path: Path to the markdown file to process (relative recommended)
87+
content_root: Root path of content directory (defaults to ./content)
88+
89+
Returns:
90+
True if successful, False otherwise
91+
"""
92+
file_path_obj = Path(file_path)
93+
94+
# Warn if absolute paths are supplied
95+
if file_path_obj.is_absolute():
96+
print(
97+
f"Warning: Using absolute path {file_path_obj} (relative paths recommended)"
98+
)
99+
100+
if not file_path_obj.exists():
101+
print(f"Error: File not found: {file_path_obj}")
102+
return False
103+
104+
if file_path_obj.suffix != ".md":
105+
print(f"Warning: File does not have .md extension: {file_path_obj}")
106+
107+
# Find content root
108+
content_root_obj: Path
109+
if content_root is None:
110+
# First check relative to current directory
111+
if Path("content").exists():
112+
content_root_obj = Path("content").resolve()
113+
else:
114+
# Walk up from file to find content directory
115+
current = file_path_obj.parent.resolve()
116+
found_root = None
117+
while current.parent != current: # Stop at filesystem root
118+
if (current / "content").exists():
119+
found_root = current / "content"
120+
break
121+
current = current.parent
122+
123+
if found_root is None:
124+
print("Error: Could not find content/ directory")
125+
return False
126+
content_root_obj = found_root
127+
else:
128+
content_root_obj = Path(content_root)
129+
if content_root_obj.is_absolute():
130+
print(
131+
f"Warning: Using absolute content root {content_root_obj} "
132+
+ "(relative paths recommended)"
133+
)
134+
135+
# Read file
136+
with open(file_path_obj, "r", encoding="utf-8") as f:
137+
content = f.read()
138+
139+
original_content = content
140+
conversions: list[tuple[str, str]] = []
141+
142+
# Find all matrix.org links
143+
for match in re.finditer(
144+
r"https://matrix\.org(/[^)>\s#]*)(?:#([^)>\s]*))?", content
145+
):
146+
full_url = match.group(0)
147+
path = match.group(1)
148+
anchor = match.group(2)
149+
150+
# Remove leading and trailing slashes and normalize
151+
path_normalized = path.strip("/")
152+
153+
# Find the actual file
154+
resolved_path = find_file_in_content(path_normalized, content_root_obj)
155+
156+
if resolved_path:
157+
# Add anchor back if present (though this may cause validation errors)
158+
if anchor:
159+
resolved_link = resolved_path.rstrip("/") + f"#{anchor}"
160+
else:
161+
resolved_link = resolved_path
162+
163+
conversions.append((full_url, resolved_link))
164+
content = content.replace(full_url, resolved_link)
165+
else:
166+
print(f"Warning: Could not resolve: {full_url}")
167+
168+
# Write file if changed
169+
if content != original_content:
170+
bytes_written = 0
171+
with open(file_path_obj, "w", encoding="utf-8") as f:
172+
bytes_written = f.write(content)
173+
if bytes_written == 0:
174+
print(f"Warning: No bytes written to {file_path_obj}")
175+
176+
print(f"✓ Converted {len(conversions)} links in {file_path_obj}")
177+
for old, new in conversions:
178+
print(f" {old}")
179+
print(f" → {new}")
180+
return True
181+
else:
182+
print(f"No matrix.org links found in {file_path_obj}")
183+
return True
184+
185+
186+
def main():
187+
if len(sys.argv) < 2:
188+
print(__doc__)
189+
sys.exit(1)
190+
191+
file_path = sys.argv[1]
192+
content_root = sys.argv[2] if len(sys.argv) > 2 else None
193+
194+
success = convert_matrix_org_links(file_path, content_root)
195+
sys.exit(0 if success else 1)
196+
197+
198+
if __name__ == "__main__":
199+
main()

0 commit comments

Comments
 (0)