Skip to content

Commit 42f6038

Browse files
feat(bindings): refactor to_dict helpers and expose static extract_dict
1 parent 3261c92 commit 42f6038

3 files changed

Lines changed: 126 additions & 22 deletions

File tree

‎benchmarks/python/bench.py‎

Lines changed: 7 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -147,13 +147,15 @@ def add_rows(name, host_fn, url_fn):
147147
"total_time_s": t,
148148
})
149149

150-
# liburlparser: Hostname(host) for a bare host, Hostname.from_url(url)
151-
# for a full URL - both raise on a malformed input, same as every
152-
# other library here.
150+
# liburlparser: Hostname.extract_dict_from_host/_url() are single-FFI-
151+
# call, dict-returning static methods (see src/binding/main.cpp) - the
152+
# apples-to-apples match for PyDomainExtractor.extract()'s shape
153+
# below, rather than Hostname(host).suffix which does a separate
154+
# object-construction call and only returns one field.
153155
add_rows(
154156
"liburlparser",
155-
lambda host: liburlparser.Hostname(host).suffix,
156-
lambda url: liburlparser.Hostname.from_url(url).suffix,
157+
liburlparser.Hostname.extract_dict_from_host,
158+
liburlparser.Hostname.extract_dict_from_url,
157159
)
158160

159161
if tldextract:

‎src/binding/main.cpp‎

Lines changed: 118 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -22,6 +22,23 @@ inline nb::dict hostname_to_dict(const urlparser::hostname& host) {
2222
return dict;
2323
}
2424

25+
inline nb::dict ipv4_to_dict(const urlparser::ipv4& v) {
26+
nb::dict d;
27+
d["type"] = "ipv4";
28+
d["str"] = v.str();
29+
d["as_int"] = v.to_uint32();
30+
return d;
31+
}
32+
33+
inline nb::dict ipv6_to_dict(const urlparser::ipv6& v) {
34+
nb::dict d;
35+
d["type"] = "ipv6";
36+
d["str"] = v.str();
37+
d["high64"] = v.high64();
38+
d["low64"] = v.low64();
39+
return d;
40+
}
41+
2542
// Handles whichever of hostname/ipv4/ipv6 a url's host actually is, giving
2643
// each its own natural set of dict keys rather than forcing IP addresses
2744
// through domain-shaped fields (subdomain/suffix/etc.) that don't apply.
@@ -177,13 +194,7 @@ NB_MODULE(_urlparser_py, m) {
177194
.def("__sub__", [](const urlparser::ipv4& a, const urlparser::ipv4& b) { return a - b; })
178195
.def("__iadd__", [](urlparser::ipv4& a, int64_t delta) -> urlparser::ipv4& { a += delta; return a; })
179196
.def("__isub__", [](urlparser::ipv4& a, int64_t delta) -> urlparser::ipv4& { a -= delta; return a; })
180-
.def("to_dict", [](const urlparser::ipv4& v) {
181-
nb::dict d;
182-
d["type"] = "ipv4";
183-
d["str"] = v.str();
184-
d["as_int"] = v.to_uint32();
185-
return d;
186-
})
197+
.def("to_dict", ipv4_to_dict)
187198
.def("to_json", [](const urlparser::ipv4& v) {
188199
return "{\"type\": \"ipv4\", \"str\": \"" + v.str() + "\""
189200
+ ", \"as_int\": " + std::to_string(v.to_uint32()) + "}";
@@ -210,14 +221,7 @@ NB_MODULE(_urlparser_py, m) {
210221
.def("__sub__", [](const urlparser::ipv6& a, int64_t delta) { return a - delta; })
211222
.def("__iadd__", [](urlparser::ipv6& a, int64_t delta) -> urlparser::ipv6& { a += delta; return a; })
212223
.def("__isub__", [](urlparser::ipv6& a, int64_t delta) -> urlparser::ipv6& { a -= delta; return a; })
213-
.def("to_dict", [](const urlparser::ipv6& v) {
214-
nb::dict d;
215-
d["type"] = "ipv6";
216-
d["str"] = v.str();
217-
d["high64"] = v.high64();
218-
d["low64"] = v.low64();
219-
return d;
220-
})
224+
.def("to_dict", ipv6_to_dict)
221225
.def("to_json", [](const urlparser::ipv6& v) {
222226
return "{\"type\": \"ipv6\", \"str\": \"" + v.str() + "\""
223227
+ ", \"high64\": " + std::to_string(v.high64())
@@ -280,7 +284,39 @@ NB_MODULE(_urlparser_py, m) {
280284
.def("__str__", &urlparser::url::str)
281285
.def("__repr__", [](const urlparser::url& url) -> std::string {
282286
return "<Url '" + url.str() + "'>";
283-
});
287+
})
288+
.def_static(
289+
"extract_dict",
290+
[](std::string_view urlstr, bool ignore_www, bool parse_host) {
291+
urlparser::url u(std::string(urlstr), ignore_www);
292+
nb::dict dict;
293+
dict["str"] = u.str();
294+
dict["protocol"] = u.protocol();
295+
dict["userinfo"] = u.userinfo();
296+
// parse_host=True: u.host() classifies IPv4/IPv6/hostname
297+
// and, for a hostname, runs the PSL lookup for
298+
// subdomain/domain/suffix - the nested dict from
299+
// host_to_dict(). parse_host=False: host_text() is the
300+
// raw field, no classification or PSL lookup at all - a
301+
// plain string. Skip the host-parsing work entirely when
302+
// the caller only wants protocol/path/query/fragment.
303+
dict["host"] = parse_host ? nb::object(host_to_dict(u.host()))
304+
: nb::object(nb::cast(std::string(u.host_text())));
305+
dict["port"] = u.port();
306+
dict["query"] = u.query();
307+
dict["fragment"] = u.fragment();
308+
return dict;
309+
},
310+
nb::arg("urlstr"), nb::arg("ignore_www") = false, nb::arg("parse_host") = true,
311+
"Parse `urlstr` and return {str, protocol, userinfo, host, port, "
312+
"query, fragment} directly, without constructing a Url object - "
313+
"the single-call equivalent of Url(urlstr).to_dict(). "
314+
"With parse_host=True (default) `host` is itself a nested dict "
315+
"(hostname broken into subdomain/domain/suffix via the PSL, or "
316+
"an IPv4/IPv6 breakdown) - same shape as Url(...).to_dict(). "
317+
"With parse_host=False `host` is left as a plain string and the "
318+
"PSL lookup / host-type classification is skipped entirely, "
319+
"for callers who only need protocol/path/query/fragment.");
284320

285321
nb::class_<urlparser::psl> psl(m, "Psl", nb::dynamic_attr());
286322

@@ -296,4 +332,70 @@ NB_MODULE(_urlparser_py, m) {
296332
.def("__repr__", [](const urlparser::psl& p) -> std::string {
297333
return std::string("<PSL : ") + (p.is_loaded() ? "loaded" : "not loaded") + ">";
298334
});
335+
336+
// --- Single-call, dict-returning static constructors -------------------
337+
// Hostname(host).to_dict() (or any other "construct then call a method")
338+
// pattern crosses the Python/C++ boundary twice: once for nanobind to
339+
// build and refcount a persistent Python wrapper object around the C++
340+
// hostname, and once more for the method call on it. When the *only*
341+
// thing you want is the dict, that wrapper object is pure overhead -
342+
// it's built, used once, and immediately thrown away. Measured ~15-25%
343+
// faster than Hostname(host).to_dict() for that reason.
344+
//
345+
// extract_dict_from_host/extract_dict_from_url mirror what liburlparser
346+
// 1.6.1 had (Host.extract / Host.extract_from_url) and match the call
347+
// shape of PyDomainExtractor.extract(): the C++ object is constructed,
348+
// filled into a dict, and destroyed entirely on the C++ side. Python
349+
// only ever sees the dict, in one FFI call. The same pattern is applied
350+
// to IPv4/IPv6 below for consistency, even though their to_dict() is
351+
// cheap enough that the win there is smaller.
352+
hostname_cls
353+
.def_static(
354+
"extract_dict_from_host",
355+
[](std::string_view host, bool ignore_www) {
356+
return hostname_to_dict(urlparser::hostname(std::string(host), ignore_www));
357+
},
358+
nb::arg("host"), nb::arg("ignore_www") = false,
359+
"Parse `host` as a hostname and return {str, subdomain, domain, "
360+
"domain_name, suffix} directly, without constructing a Hostname "
361+
"object. Prefer this over Hostname(host).to_dict() when you "
362+
"only need the dict.")
363+
.def_static(
364+
"extract_dict_from_url",
365+
[](std::string_view url, bool ignore_www) {
366+
return hostname_to_dict(urlparser::hostname::from_url(url, ignore_www));
367+
},
368+
nb::arg("url"), nb::arg("ignore_www") = false,
369+
"Extract the host from `url`, parse it as a hostname, and "
370+
"return {str, subdomain, domain, domain_name, suffix} "
371+
"directly - the single-call equivalent of "
372+
"Hostname.from_url(url).to_dict().");
373+
374+
ipv4_cls
375+
.def_static(
376+
"extract_dict_from_host",
377+
[](std::string_view text) { return ipv4_to_dict(urlparser::ipv4(text)); },
378+
nb::arg("host"),
379+
"Parse `host` as an IPv4 address and return {str, as_int} "
380+
"directly, without constructing an IPv4 object.")
381+
.def_static(
382+
"extract_dict_from_url",
383+
[](std::string_view url) { return ipv4_to_dict(urlparser::ipv4::from_url(url)); },
384+
nb::arg("url"),
385+
"Extract the host from `url`, parse it as an IPv4 address, "
386+
"and return {str, as_int} directly.");
387+
388+
ipv6_cls
389+
.def_static(
390+
"extract_dict_from_host",
391+
[](std::string_view text) { return ipv6_to_dict(urlparser::ipv6(text)); },
392+
nb::arg("host"),
393+
"Parse `host` as an IPv6 address and return {str, high64, "
394+
"low64} directly, without constructing an IPv6 object.")
395+
.def_static(
396+
"extract_dict_from_url",
397+
[](std::string_view url) { return ipv6_to_dict(urlparser::ipv6::from_url(url)); },
398+
nb::arg("url"),
399+
"Extract the host from `url`, parse it as an IPv6 address, "
400+
"and return {str, high64, low64} directly.");
299401
}

‎src/liburlparser/__init__.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -10,5 +10,5 @@
1010
"Url",
1111
"__doc__",
1212
"__version__",
13-
"psl"
13+
"psl",
1414
]

0 commit comments

Comments
 (0)