Repository navigation
Expand file tree
/
Copy pathprocess_langsmith_openapi.py
More file actions
executable file
·357 lines (307 loc) · 10.7 KB
/
Copy pathprocess_langsmith_openapi.py
File metadata and controls
executable file
·357 lines (307 loc) · 10.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
#!/usr/bin/env python3
"""Post-process the LangSmith OpenAPI spec for public documentation.
Adds ``x-hidden: true`` to fleet and internal endpoints so Mintlify skips them,
and injects ``x-group`` on tags so auto-generated pages are grouped under
human-readable headings.
Usage
-----
Fetch the live spec from api.smith.langchain.com and write to src/::
python scripts/process_langsmith_openapi.py --write
Read from a local file instead::
python scripts/process_langsmith_openapi.py --input /path/to/openapi.json --write
Preview to stdout (dry run)::
python scripts/process_langsmith_openapi.py --input /path/to/openapi.json
"""
from __future__ import annotations
import argparse
import json
import ssl
import sys
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
# ---------------------------------------------------------------------------
# Constants
# ---------------------------------------------------------------------------
REPO_ROOT = Path(__file__).resolve().parent.parent
OUTPUT_PATH = REPO_ROOT / "src" / "langsmith" / "langsmith-platform-openapi.json"
# Only these hosts are allowed when fetching a spec over the network.
ALLOWED_HOSTS = {
"api.smith.langchain.com",
}
DEFAULT_URL = "https://api.smith.langchain.com/openapi.json"
FETCH_TIMEOUT_SECONDS = 30
# Tags whose operations should be hidden from the public docs.
HIDDEN_TAGS: set[str] = {
# Fleet
"agents",
"fleet auth",
"fleet credentials",
"fleet github-app",
"fleet integrations",
"fleet mcp",
"fleet threads",
"fleet trigger-templates",
"fleet triggers",
"fleet usage",
"fleet_webhooks",
"skills",
# Internal / infra
"beacon",
"nps",
"sandboxes-internal",
"internal",
"admin-panel",
"provisioning",
"debug",
# Low-value system endpoints
"metrics",
# Untagged health checks are caught by HIDDEN_PATHS below.
}
# Exact paths that should always be hidden (health checks, etc.).
HIDDEN_PATHS: set[str] = {
"/api/v1/ok",
"/ok",
}
# Path prefixes that should always be hidden regardless of tag.
HIDDEN_PATH_PREFIXES: list[str] = [
"/v1/fleet/",
"/v1/platform/fleet/",
"/v1/platform/fleet-webhooks/",
"/v1/beacon/",
"/v1/platform/nps/",
"/v2/sandboxes/internal/",
]
# Map raw tag names to human-readable group headings (``x-group``).
# Tags not listed here keep their original name as the group heading.
TAG_GROUPS: dict[str, str] = {
# Tracing
"run": "Tracing",
"runs": "Tracing",
"tracer-sessions": "Tracing",
"sessions": "Tracing",
# Datasets
"datasets": "Datasets",
"examples": "Datasets",
# Evaluation
"experiments": "Evaluation",
"evaluators": "Evaluation",
"experiment-view-overrides": "Evaluation",
# Feedback & annotation
"feedback": "Feedback & Annotation",
"feedback-configs": "Feedback & Annotation",
"annotation-queues": "Feedback & Annotation",
"annotation_queues": "Feedback & Annotation",
# Prompts & playground
"prompts": "Prompts & Playground",
"prompt-webhooks": "Prompts & Playground",
"playground-settings": "Prompts & Playground",
"commits": "Prompts & Playground",
"directories": "Prompts & Playground",
"hub_environments": "Prompts & Playground",
"tag-transitions": "Prompts & Playground",
# Prompt hub
"repos": "Prompt Hub",
"comments": "Prompt Hub",
"likes": "Prompt Hub",
"tags": "Prompt Hub",
"ownerships": "Prompt Hub",
"settings": "Prompt Hub",
"optimization-jobs": "Prompt Hub",
# Monitoring
"charts": "Monitoring",
"alert_rules": "Monitoring",
"bulk-exports": "Monitoring",
# Sandboxes
"sandboxes": "Sandboxes",
"checkpoint": "Sandboxes",
"execute": "Sandboxes",
# Administration
"Organizations": "Administration",
"orgs": "Administration",
"workspaces": "Administration",
"tenant": "Administration",
"api-key": "Administration",
"auth": "Administration",
"me": "Administration",
"service-accounts": "Administration",
"SCIM Tokens": "Administration",
"TTL Settings": "Administration",
"ttl-settings": "Administration",
"access_policies": "Administration",
"audit-logs": "Administration",
"usage-limits": "Administration",
"data_planes": "Administration",
"aws_marketplace": "Administration",
# LLM Gateway
"gateway-policies": "LLM Gateway",
# Integrations & tools
"integrations": "Integrations & Tools",
"tools": "Integrations & Tools",
"mcp_vendors": "Integrations & Tools",
"mcp": "Integrations & Tools",
"oauth": "Integrations & Tools",
# Issues
"issues": "Issues",
"issues-agent": "Issues",
# Files
"files": "Files",
# System
"info": "System",
"features": "System",
"model-price-map": "System",
"public": "System",
"ace": "System",
"backfills": "System",
"threads": "System",
}
# Display order for groups in the generated docs sidebar.
# Groups not listed here are appended alphabetically after the listed ones.
GROUP_ORDER: list[str] = [
"Tracing",
"Datasets",
"Evaluation",
"Feedback & Annotation",
"Monitoring",
"Prompts & Playground",
"Prompt Hub",
"Integrations & Tools",
"LLM Gateway",
"Sandboxes",
"Issues",
"Administration",
"Files",
"System",
]
# ---------------------------------------------------------------------------
# Processing
# ---------------------------------------------------------------------------
def _should_hide_by_path(path: str) -> bool:
"""Return True if the path matches a hidden prefix or exact path."""
return (
path in HIDDEN_PATHS
or any(path.startswith(prefix) for prefix in HIDDEN_PATH_PREFIXES)
)
def _should_hide_by_tags(tags: list[str]) -> bool:
"""Return True if any of the operation's tags are in the hidden set."""
return any(tag in HIDDEN_TAGS for tag in tags)
def process_spec(spec: dict) -> dict:
"""Add ``x-hidden`` and ``x-group`` annotations to *spec* in place."""
hidden_count = 0
total_count = 0
# 1. Mark operations as hidden.
for path, methods in spec.get("paths", {}).items():
for method, operation in methods.items():
if not isinstance(operation, dict):
continue
# Skip OpenAPI metadata keys like "parameters", "summary" at path level.
if method in ("parameters", "summary", "description", "servers"):
continue
total_count += 1
tags = operation.get("tags", [])
if _should_hide_by_path(path) or _should_hide_by_tags(tags):
operation["x-hidden"] = True
hidden_count += 1
# 2. Ensure top-level tags array exists and add x-group.
if "tags" not in spec:
spec["tags"] = []
existing_names = {t["name"] for t in spec["tags"]}
# Update existing tag objects.
for tag_obj in spec["tags"]:
name = tag_obj["name"]
if name in TAG_GROUPS:
tag_obj["x-group"] = TAG_GROUPS[name]
# Also hide top-level tag entries for hidden tags.
if name in HIDDEN_TAGS:
tag_obj["x-hidden"] = True
# Add tag objects for tags that appear only on operations.
seen_on_operations: set[str] = set()
for _path, methods in spec.get("paths", {}).items():
for _method, operation in methods.items():
if isinstance(operation, dict):
for tag in operation.get("tags", []):
seen_on_operations.add(tag)
for tag_name in sorted(seen_on_operations - existing_names):
entry: dict = {"name": tag_name}
if tag_name in TAG_GROUPS:
entry["x-group"] = TAG_GROUPS[tag_name]
if tag_name in HIDDEN_TAGS:
entry["x-hidden"] = True
spec["tags"].append(entry)
# 3. Sort tags so groups appear in GROUP_ORDER.
group_rank = {g: i for i, g in enumerate(GROUP_ORDER)}
fallback = len(GROUP_ORDER)
def _tag_sort_key(tag_obj: dict) -> tuple:
group = tag_obj.get("x-group", tag_obj["name"])
return (group_rank.get(group, fallback), group, tag_obj["name"])
spec["tags"] = sorted(spec["tags"], key=_tag_sort_key)
print(
f"Processed {total_count} operations: "
f"{hidden_count} hidden, {total_count - hidden_count} public",
file=sys.stderr,
)
return spec
# ---------------------------------------------------------------------------
# I/O
# ---------------------------------------------------------------------------
def fetch_spec(url: str) -> dict:
"""Fetch an OpenAPI JSON spec from *url*."""
parsed = urllib.parse.urlparse(url)
if parsed.hostname not in ALLOWED_HOSTS:
raise ValueError(
f"Host {parsed.hostname!r} is not in the allow-list: {ALLOWED_HOSTS}"
)
ctx = ssl.create_default_context()
req = urllib.request.Request(url, headers={"Accept": "application/json"})
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT_SECONDS, context=ctx) as resp:
return json.loads(resp.read())
def load_spec(path: str) -> dict:
"""Load an OpenAPI JSON spec from a local file."""
resolved = Path(path).resolve()
return json.loads(resolved.read_text(encoding="utf-8"))
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
def main() -> None:
parser = argparse.ArgumentParser(
description="Post-process the LangSmith OpenAPI spec for public docs."
)
parser.add_argument(
"--input",
help=(
"Path to a local OpenAPI JSON file. "
f"If omitted, fetches from {DEFAULT_URL}."
),
)
parser.add_argument(
"--output",
help=f"Output path (default: {OUTPUT_PATH.relative_to(REPO_ROOT)})",
default=str(OUTPUT_PATH),
)
parser.add_argument(
"--write",
action="store_true",
help="Write to output file. Without this flag, prints to stdout.",
)
args = parser.parse_args()
# Load.
if args.input:
print(f"Loading spec from {args.input}", file=sys.stderr)
spec = load_spec(args.input)
else:
print(f"Fetching spec from {DEFAULT_URL}", file=sys.stderr)
spec = fetch_spec(DEFAULT_URL)
# Process.
result = process_spec(spec)
output_json = json.dumps(result, indent=2, ensure_ascii=False) + "\n"
# Write.
if args.write:
out = Path(args.output).resolve()
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(output_json, encoding="utf-8")
print(f"Wrote {len(output_json):,} bytes to {out}", file=sys.stderr)
else:
print(output_json)
if __name__ == "__main__":
main()