Repository navigation
Expand file tree
/
Copy pathurl_utils.py
More file actions
249 lines (201 loc) · 8.71 KB
/
Copy pathurl_utils.py
File metadata and controls
249 lines (201 loc) · 8.71 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
# WriterAgent - AI Writing Assistant for LibreOffice
# Copyright (c) 2026 KeithCu
#
# SPDX-License-Identifier: GPL-3.0-or-later
"""
URL parsing utilities for WriterAgent.
HTTP / LLM endpoint helpers live below. Filesystem ``file:`` conversion is a
separate section — do not fold it into ``normalize_endpoint_url``.
"""
# crosshair: off
from __future__ import annotations
import os
import urllib.parse
from pathlib import Path
from typing import Any
from plugin.framework.constants import EXTENSION_ID_LIBREPY
from plugin.framework.deal_shim import DEAL_MAX_URL, UNDER_CROSSHAIR, ascii_bounded, deal
def _deal_url_ok_pytest(url: object) -> bool:
# Endpoint URLs with query strings exceed DEAL_MAX_URL, and IRI hosts
# are not ASCII. The cap raised PreContractError before normalize
# returned "". The body treats a non-str as empty. CrossHair keeps
# the short ASCII domain. ``url`` is unused on this profile.
return True
def _deal_url_ok_crosshair(url: object) -> bool:
return url is None or ascii_bounded(url, DEAL_MAX_URL)
_deal_url_ok = _deal_url_ok_crosshair if UNDER_CROSSHAIR else _deal_url_ok_pytest
LIBREPY_DISPATCH_PROTOCOL = EXTENSION_ID_LIBREPY + ":"
def matches_librepy_dispatch_url(url: Any) -> bool:
"""Return True when *url* is a LibrePy menu/protocol dispatch URL."""
proto = str(getattr(url, "Protocol", None) or "")
if proto.startswith(EXTENSION_ID_LIBREPY):
return True
complete = str(getattr(url, "Complete", None) or "")
return complete.startswith(LIBREPY_DISPATCH_PROTOCOL)
def dispatch_command_from_url(url: Any, *, protocol_prefix: str = LIBREPY_DISPATCH_PROTOCOL) -> str:
"""Extract handler command from a LibreOffice dispatch URL.
LO usually sets ``Path`` (e.g. ``main.settings``). Some dispatch paths only
populate ``Complete`` (``org.extension.librepy:main.settings``); derive Path
from that so menu handlers still run.
"""
path = str(getattr(url, "Path", None) or "").strip().lstrip("/")
if path:
return path
complete = str(getattr(url, "Complete", None) or "").strip()
if complete.startswith(protocol_prefix):
return complete[len(protocol_prefix) :].lstrip("/")
if ":" in complete:
return complete.split(":", 1)[1].lstrip("/")
return complete
def _is_zai_host(url: Any) -> bool:
"""True when URL targets Z.ai (general or coding-plan API)."""
host = get_url_hostname(url).lower()
return host == "z.ai" or host.endswith(".z.ai")
def _is_google_host(url: Any) -> bool:
"""True when URL targets Google Gemini API (generativelanguage.googleapis.com)."""
host = get_url_hostname(url).lower()
name = "generativelanguage.googleapis.com"
return host == name or host.endswith("." + name)
def _zai_url_path(url: Any) -> str:
"""Normalized path without trailing slash (empty string when bare host)."""
if not isinstance(url, str):
return ""
return (urllib.parse.urlparse(url).path or "").rstrip("/")
@deal.pre(lambda url, is_openwebui=False: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str) and result.startswith("/"))
def get_api_version_suffix(url: Any, is_openwebui: bool = False) -> str:
"""Return the API version suffix (e.g. '/v1', '/v4', '/api/paas/v4') for a given endpoint URL."""
if is_openwebui:
return "/api"
# Z.ai: bare host uses general OpenAI base (/api/paas/v4); deeper paths append /v4 only.
if _is_zai_host(url):
if _zai_url_path(url) in ("", "/"):
return "/api/paas/v4"
return "/v4"
# Google Gemini: OpenAI-compatible API base is /v1beta/openai
if _is_google_host(url):
return "/v1beta/openai"
return "/v1"
@deal.pre(lambda url, is_openwebui=False: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str))
@deal.ensure(lambda url, is_openwebui=False, result="": bool(isinstance(url, str) and url.strip()) or result == "")
def normalize_endpoint_url(url: Any, is_openwebui: bool = False) -> str:
"""Clean up endpoint URL: strip whitespace, trailing slashes, and domain-specific version suffixes."""
if type(url) is not str or not url.strip():
return ""
url = url.strip()
# Remove trailing /
while url.endswith("/"):
url = url[:-1]
# Open WebUI chat is {base}/api/chat/completions — strip pasted /api/v1, /api, or /v1 in one pass.
# (Half-stripping /api/v1 → /api then appending /api again yields /api/api/chat/completions.)
if is_openwebui:
lower = url.lower()
if lower.endswith("/api/v1"):
return url[: -len("/api/v1")]
if lower.endswith("/api"):
return url[: -len("/api")]
if lower.endswith("/v1"):
return url[:-3]
return url
if _is_google_host(url):
lower = url.lower()
if lower.endswith("/v1beta/openai"):
return url[: -len("/v1beta/openai")]
if lower.endswith("/v1beta"):
return url[: -len("/v1beta")]
if lower.endswith("/v1"):
return url[:-3]
return url
# Remove the version suffix we expect to add back (e.g. /v1, /v4, /api/paas/v4)
suffix = get_api_version_suffix(url, is_openwebui=False)
if url.lower().endswith(suffix):
url = url[: -len(suffix)]
elif _is_zai_host(url) and url.lower().endswith("/v4"):
# Legacy preset stored https://api.z.ai/v4 before general base was /api/paas/v4.
url = url[:-3]
elif url.lower().endswith("/v1"):
# Always strip /v1 as a fallback for custom endpoints
url = url[:-3]
return url
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str))
def get_url_hostname(url: Any) -> str:
"""Return hostname from URL safely."""
if type(url) is not str:
return ""
try:
parsed = urllib.parse.urlparse(url)
return parsed.hostname or ""
except (ValueError, TypeError, AttributeError):
# urlparse rejects non-str (TypeError); keep "safely" for CrossHair/fuzz inputs.
return ""
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str))
def get_url_domain(url: Any) -> str:
"""Return 'example.com' from 'https://api.example.com/v1'."""
host = get_url_hostname(url)
if not host:
return ""
parts = host.split(".")
if len(parts) >= 2:
return ".".join(parts[-2:])
return host
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str))
def get_url_path(url: Any) -> str:
"""Return path from URL safely."""
if type(url) is not str:
return ""
try:
parsed = urllib.parse.urlparse(url)
return parsed.path or ""
except (ValueError, TypeError, AttributeError):
return ""
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, dict))
def get_url_query_dict(url: Any) -> dict[str, list[str]]:
"""Return query parameters as dict (values are lists)."""
if type(url) is not str or not url:
return {}
try:
parsed = urllib.parse.urlparse(url)
return urllib.parse.parse_qs(parsed.query)
except (ValueError, TypeError, AttributeError):
return {}
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, str))
def get_url_path_and_query(url: Any) -> str:
"""Return path + query string from URL."""
if type(url) is not str:
return "/"
try:
parsed = urllib.parse.urlparse(url)
path = parsed.path or "/"
if parsed.query:
return f"{path}?{parsed.query}"
return path
except (ValueError, TypeError, AttributeError):
return "/"
@deal.pre(lambda url: _deal_url_ok(url))
@deal.post(lambda result: isinstance(result, bool))
def is_pdf_url(url: Any) -> bool:
"""Check for .pdf in the URL path safely."""
if type(url) is not str:
return False
try:
parsed = urllib.parse.urlparse(url)
return (parsed.path or "").lower().endswith(".pdf")
except (ValueError, TypeError, AttributeError):
return False
# ---------------------------------------------------------------------------
# Filesystem / document file: URLs (UNO-free; not HTTP endpoints)
# ---------------------------------------------------------------------------
def path_to_file_url(path: str) -> str:
"""Build a document / ``loadComponentFromURL`` / GraphicProvider-style ``file:`` URL.
Normalize with ``os.path.abspath`` then ``Path.as_uri()`` so Unix yields
``file:///…`` (three slashes). ``urljoin('file:', …)`` wrongly produces
``file:/…``. Not an LLM endpoint helper; no ``@deal`` (those contracts are
for HTTP strings). UNO-free so embeddings extract can import this module.
"""
return Path(os.path.abspath(path)).as_uri()