diff options
| -rw-r--r-- | .gitignore | 2 | ||||
| -rwxr-xr-x | Justfile | 4 | ||||
| -rwxr-xr-x | base.kuht | 21 | ||||
| -rwxr-xr-x | build.py3 | 63 | ||||
| -rwxr-xr-x | index.html | 36 | ||||
| -rwxr-xr-x | src/base.kuht | 22 | ||||
| -rwxr-xr-x | src/blog/an-inheritance-detox.kuht | 853 | ||||
| -rwxr-xr-x | src/blog/best-tool-for-the-job.kuht | 685 | ||||
| -rw-r--r-- | src/blog/color-spaces.kuht | 130 | ||||
| -rwxr-xr-x | src/blog/comments-at-end.kuht | 662 | ||||
| -rw-r--r-- | src/blog/datetime.kuht | 518 | ||||
| -rwxr-xr-x | src/blog/how-happylock-works.kuht | 767 | ||||
| -rw-r--r-- | src/blog/linking.kuht | 245 | ||||
| -rw-r--r-- | src/blog/named-optional-args.kuht | 1701 | ||||
| -rw-r--r-- | src/blog/quality-quantity.kuht | 98 | ||||
| -rwxr-xr-x | src/blog/try-catch-in-rust.kuht | 553 | ||||
| -rwxr-xr-x | src/blog/unwind-safety-happylock.kuht | 555 | ||||
| -rwxr-xr-x | src/index.kuht | 33 | ||||
| -rwxr-xr-x | style-dark.css | 22 | ||||
| -rwxr-xr-x | style-light.css | 22 | ||||
| -rwxr-xr-x | style.css | 102 |
21 files changed, 7094 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..ce1d3ee --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +target/** +.ruff_cache/** diff --git a/Justfile b/Justfile new file mode 100755 index 0000000..e32ea1a --- /dev/null +++ b/Justfile @@ -0,0 +1,4 @@ +sync: + python3 build.py3 + rsync -azvP 'target/' botahamec@botahamec.dev:/var/www/botahamec.dev + rsync -azvP 'target/' botahamec@altahamec.dev:/var/www/botahamec.dev diff --git a/base.kuht b/base.kuht new file mode 100755 index 0000000..bafbfe6 --- /dev/null +++ b/base.kuht @@ -0,0 +1,21 @@ +<head> + <meta name="viewport" content="width=device-width, initial-scale=1" /> + + <link href="/style.css" rel="stylesheet" /> + <link href="/style-light.css" rel="stylesheet" title="Default Style" /> + <link href="/style-dark.css" rel="alternate stylesheet" title="Dark Mode" /> +</head> + +<block "navbar"> + <header class="nav"> + <a class="nav__button screen-reader-focusable" href="#article">Skip Navigation</a> + <h1 class="nav__title">Botahamec</h1> + <nav class="nav__menu"> + <menu class="nav__menu"> + <li class="nav__item"><a class="nav__button" href="/">Home</a></li> + <li class="nav__item"><a class="nav__button" href="/blog">Blog</a></li> + <li class="nav__item"><a class="nav__button" href="/projects">Projects</a></li> + </menu> + </nav> + </header> +</block> diff --git a/build.py3 b/build.py3 new file mode 100755 index 0000000..03c4cb5 --- /dev/null +++ b/build.py3 @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- + +import brotli +import gzip +import os +import shutil + + +def clean_target(): + if os.path.exists('target'): + shutil.rmtree('target') + os.mkdir('target') + + +def compile_all(path: str, target: str): + for file in os.listdir(path): + src = os.path.join(path, file) + base, ext = os.path.splitext(file) + if os.path.isdir(src): + os.mkdir(os.path.join(target, file)) + compile_all(src, os.path.join(target, file)) + elif ext == '.kuht': + os.system(f"kuht {src} src/base.kuht -o {target}/{base}.html") + + +def move_css(path: str): + for file in os.listdir(path): + if os.path.splitext(file)[1] == '.css': + shutil.copy(file, os.path.join('target', file)) + + +def compress_files(path: str): + for file in os.listdir(path): + file = os.path.join(path, file) + if os.path.isdir(file): + compress_files(file) + return + + base, ext = os.path.splitext(file) + with open(file, 'rb') as f_in: + data = f_in.read() + with gzip.open(f"{file}.gz", 'wb') as f_out: + f_out.write(data) + with open(f"{file}.br", 'wb') as f_out: + f_out.write(brotli.compress(data, mode=brotli.MODE_TEXT)) + + +def main(): + original_path = os.getcwd() + script_path = os.path.dirname(os.path.realpath(__file__)) + os.chdir(script_path) + + clean_target() + compile_all('src', 'target') + move_css(".") + compress_files("target") + + os.chdir(original_path) + + +if __name__ == '__main__': + main() diff --git a/index.html b/index.html new file mode 100755 index 0000000..0ce371f --- /dev/null +++ b/index.html @@ -0,0 +1,36 @@ +<!DOCTYPE html> +<html lang="en"> + <head> + <title>Botahamec's Website</title> + <link href="/style.css" rel="stylesheet" /> + <link href="/style-light.css" rel="stylesheet" title="Default Style" /> + <link href="/style-dark.css" rel="alternate stylesheet" title="Dark Mode" /> + <meta name="viewport" content="width=device-width, initial-scale=1"> + <meta name="description" content="All of Bota's projects and endeavors" /> + <meta charset="utf-8" /> + </head> + <body> + <header><h1>Botahamec</h1></header> + + <p> + Hi. This is a website I threw together quickly so that I could write my blog post. It's not + much right now. But you should read my blog posts anyway. + </p> + + <h2>Projects</h2> + <ul> + <li><a href="https://www.github.com/botahamec/happylock">HappyLock</a></li> + <li><a href="https://www.github.com/botahamec/exun">Exun</a></li> + <li><a href="https://www.github.com/botahamec/snob">Snob</a></li> + </ul> + + <h2>Blog</h2> + <ul> + <li><a href="/blog/how-happylock-works.html">How HappyLock Works</a></li> + <li><a href="/blog/best-tool-for-the-job.html">The Best Tool for the Job</a></li> + <li><a href="/blog/an-inheritance-detox.html">An Inheritance Detox</a></li> + <li><a href="/blog/try-catch-in-rust.html">Introducing try-catch to Rust</a></li> + <li><a href="/blog/comments-at-end.html">Comments at the End</a></li> + </ul> + </body> +</html> diff --git a/src/base.kuht b/src/base.kuht new file mode 100755 index 0000000..f87ea8f --- /dev/null +++ b/src/base.kuht @@ -0,0 +1,22 @@ +<head> + <meta name="viewport" content="width=device-width, initial-scale=1" /> + <meta charset="utf-8" /> + + <link href="/style.css" rel="stylesheet" /> + <link href="/style-light.css" rel="stylesheet" title="Default Style" /> + <link href="/style-dark.css" rel="alternate stylesheet" title="Dark Mode" /> +</head> + +<block "navbar"> + <header class="nav"> + <a class="nav__button screen-reader-focusable" href="#article">Skip Navigation</a> + <h1 class="nav__title">Botahamec</h1> + <nav class="nav__menu"> + <menu class="nav__menu"> + <li class="nav__item"><a class="nav__button" href="/">Home</a></li> + <li class="nav__item"><a class="nav__button" href="/blog">Blog</a></li> + <li class="nav__item"><a class="nav__button" href="/projects">Projects</a></li> + </menu> + </nav> + </header> +</block> diff --git a/src/blog/an-inheritance-detox.kuht b/src/blog/an-inheritance-detox.kuht new file mode 100755 index 0000000..200380f --- /dev/null +++ b/src/blog/an-inheritance-detox.kuht @@ -0,0 +1,853 @@ +<import "base.kuht" as "base" /> + +<head> + <title>An Inheritance Detox</title> + <meta name="description" content="I once pointed out to someone, who was learning Rust, that Rust doesn't have inheritance. Even people who like object-oriented programming don't like inheritance, so this is how I explained why and what you should do instead." /> +</head> + +<body> + +<article> + +<h1>An Inheritance Detox</h1> + +<p> + I once pointed out to someone, who was learning Rust, that Rust doesn't have inheritance. He's a + software engineering major at RIT<a id="af-1" href="#footnote-1"><sup>1</sup></a>, so this was a + shock to him. I eventually told him I would need to give him an object-oriented programming detox. + But as I was researching, I learned that even people who like object-oriented programming don't + defend inheritance anymore. Inheritance was the problem he was dealing with, so I decided to just + focus on that. This post is a blog-translation of the presentation I eventually gave to the Buffalo + Rust User Group. +</p> + +<blockquote> + I once attended a Java user group meeting where James Gosling (Java's inventor) was the featured + speaker. During the memorable Q&A session, someone asked him: "If you could do Java over again, + what would you change?" "I'd leave out classes", he replied. After the laughter died down, he + explained that the real problem wasn't classes per se, but rather implementation inheritance + (the extends relationship). Interface inheritance (the implements relationship) is preferable. You + should avoid implementation inheritance whenever possible. + <p><a href="https://www.infoworld.com/article/2160788/why-extends-is-evil.html">- Allen Holub</a></p> +</blockquote> + +<h2>The Problem of Inheritance</h2> + +<p> + For starters, anything you can do with inheritance, you can also do with composition. I'll give some + examples of that later. That, on its own, should enough to make you skeptical of inheritance, but + I'll keep going. +</p> + +<p> + One problem is that inheritance leads to unnecessary coupling. Allowing the child class to modify the + base class, and vice-versa, are going to result in more testing on your part. If you ever change the + base class, not only do you need to retest that base class, but also anything that inherits from that + base class. +</p> + +<p> + Another problem is known as the fragile-base class problem. To illustrate it, here's an example of + some code that would be difficult to write in Rust: +</p> + +<pre> +public class Queue<E> extends ArrayList<E> { + private int start; + + public Queue() { + this.start = 0; + } + + public void enqueue(E element) { + this.add(element); + } + + public E dequeue() { + return this.get(this.start++); + } +} +</pre> + +<p> + Say, for the sake of argument, that when you implemented the <code>Queue</code>, the + <code>ArrayList</code> only had two methods: <code>get</code> and <code>add</code>. Later, somebody + decides to add a <code>clear</code> method. What happens if someone calls <code>queue.clear()</code>? + Your <code>start</code> field won't be updated to reflect the new length. So you'll very quickly run + into a bug. +</p> + +<pre> +Queue<Integer> queue = new Queue<>(); +queue.enqueue(5); +queue.enqueue(6); +int x = queue.dequeue(); +// pretty normal so far + +queue.clear(); // oh no +int y = queue.dequeue(); // oh no no no no +</pre> + +<p> + Ok, so let's say, for the sake of argument, that you realize what happened and update + <code>Queue</code> quickly. +</p> + +<pre> +public class Queue<E> extends ArrayList<E> { + // ... *snip* ... + + @override + public void clear() { + this.start = 0; + super.clear(); + } +} +</pre> + +<p>That's one problem solved. But what happens when someone adds <code>removeRange</code>?</p> + +<pre> +public class Queue<E> extends ArrayList<E> { + // ... *snip* ... + + @override + public void removeRange(int from, int to) { + // ??? + } +} +</pre> + +<p> + So a better way of doing this would be to have <code>Queue</code> contain an <code>ArrayList</code>, + rather than being an <code>ArrayList</code>. +</p> + +<pre> +public class Queue<E> { + private int start; + private ArrayList<E> list; + + public Queue() { + this.start = 0; + this.list = new ArrayList<>(); + } + + public void enqueue(E element) { + this.list.add(element); + } + + public E dequeue() { + return this.list.get(this.start++); + } +} +</pre> + +<p>Note that this is actually very easy to write in Rust.</p> + +<pre> +pub struct Queue<T> { + start: int, + list: Vec<T>, +} + +impl<T> Queue<T> { + pub const fn new() -> Self { + Self { + start: 0, + list: Vec::new(), + } + } + + pub fn enqueue(&mut self, item: T) { + self.list.push(item); + } + + pub fn dequeue(&mut self) -> Option<T> { + let idx = self.start; + self.start += 1; + self.list.get(idx) + } +} +</pre> + +<p> + In Allen Holub's article, <a href="https://www.infoworld.com/article/2160788/why-extends-is-evil.html"> + Why <code>extends</code> is evil</a>, he gives another example of a problem where optimizing a + base <code>Stack</code> class causes a problem if the child classes depend on a method being called. + I won't copy and past his article, but I wanted to show that he recommended using an interface in + that case. And the interface he wrote translates very nicely to Rust. +</p> + +<pre> +pub trait Stack<T> { + fn push(&mut self, item: T); + fn pop(&mut self) -> T; + fn push_many(&mut self, items: &[T]); +} + +pub struct SimpleStack<T> { + stack_pointer: Option<usize>, + stack: Box<[T]> +} + +impl<T> Stack<T> for SimpleStack<T> { + fn push(&mut self, item: T) { + self.stack_pointer += 1; + self.stack[self.stack_pointer] = item; + } + + fn pop(&mut self) -> T { + let item = self.stack[self.stack_pointer]; + self.stack_pointer -= 1; + item + } + + fn push_many(&mut self, items: &[T]) { + // Notice that this doesn't call push. If this were a base class for + // MonitorableStack, it might incorrectly assume that this will call push. + + let start = self.stack_pointer + 1; + let end = self.stack_pointer + 1 + items.len(); + self.stack[start..end].clone_from_slice(items); + self.stack_pointer += items.len(); + } +} + +pub struct MonitorableStack<T> { + stack: SimpleStack<T>, + high_water_mark: usize, + current_size: usize, +} + +impl<T> Stack<T> for SimpleStack<T> { + fn push(&mut self, item: T) { + self.current_size += 1; + if self.current_size > self.high_water_mark { + high_water_mark = self.current_size; + } + + self.stack.push(item) + } + + fn pop(&mut self) -> T { + self.current_size -= 1; + self.stack.pop() + } + + fn push_many(&mut self, items: &[T]) { + if self.current_size + items.len() > self.high_water_mark { + self.high_water_mark = self.current_size + items.len(); + } + + // We cannot assume that our call function will be pushed, because + // that'd be impossible. We're in charge of our own destiny here. + self.stack.push_many(items) + } +} +</pre> + +<p>That was a long piece of code. But I hope that the point is clear.</p> + +<h2>Replacing Inheritance</h2> + +<p>So if we can't use inheritance, what can we use. Rust provides a few mechanisms for composition:</p> +<ul> + <li>structs (the equivalent of final class in Java)</li> + <li>traits (like interfaces in Java)</li> + <li>enums (Java has no equivalent. Java enums aren't the same.)</li> +</ul> + +<p>Let's show a few examples of common object-oriented patterns in Rust.</p> + +<h3>Factory Methods</h3> + +<p> + It's worth noting that I'm getting these patterns from <a href="https://www.oodesign.com">oodesign.com</a>. + For copyright reasons, I will link to <a href="https://www.oodesign.com/factory-method-pattern">their implementation</a>. +</p> + +<p> + Their example is actually hilarious to me, because we can replace it with a type alias, if you're ok + with monomorphization, and the fact that Rust doesn't actually care about type bounds on type alises + at the moment. +</p> + +<pre> +type Factory<P: Product> = fn() -> P; +</pre> + +<p> + The typical use-case I've seen for factories in the past is when creating the product relies on some + other state. Another case is when we can't specify the underlying type and only know that it's a + product. We can do bose of those as well. +</p> + +<pre> +trait ProductFactory { + fn create_product(&mut self) -> Box<dyn Product>; +} +</pre> + +<p>The Java version is even simpler:</p> + +<pre> +interface ProductFactory { + Product createProduct(); +} +</pre> + +<h3>Shapes</h3> + +<p> + This is a very common example of inheritance, but I think it's accidentally an example of why you + shouldn't use it. There's an example on <a href="https://www.oodesign.com/prototype-pattern">oodesign.com</a>. + Notice that calling <code>setWidth</code> on a square doesn't make very much sense. Here's what I + would do in Rust: +</p> + +<pre> +trait Shape { + fn width(&self) -> f32; + + fn height(&self) -> f32; + + fn area(&self) -> f32; + + // no need for set_width and set_height here +} + +struct Rectangle { + width: f32, + height: f32, +} + +impl Rectangle { + fn set_height(&mut self, height: f32) { + self.height = height; + } + + fn set_width(&mut self, width: f32) { + self.width = width; + } +} + +impl Shape for Rectangle { + fn width(&self) -> f32 { + self.width + } + + fn height(&self) -> f32 { + self.height + } + + fn area(&self) -> f32 { + self.width * self.height + } +} + +struct Square { + size: f32, +} + +impl Square { + fn set_size(&mut self, size: f32) { + self.size = size; + } +} + +impl Shape for Square { + fn width(&self) -> f32 { + self.size + } + + fn height(&self) -> f32 { + self.size + } + + fn area(&self) -> f32 { + self.size * self.size + } +} +</pre> + +<h3>Interpreter</h3> + +<p> + I was planning on using the Command pattern here, but after another look I noticed that the provided + example didn't even use inheritance. So instead we'll use the + <a href="https://www.oodesign.com/interpreter-pattern">interpreter pattern</a>. This pattern is + amazing for Rust, because of enums. If you don't want user code to extend the set of possible + commands, then this is really simple. +</p> + +<pre> +pub enum Expression { + Literal(Arc<str>), + Or(Arc<Expression>, Arc<Expression>), + And(Arc<Expression>, Arc<Expression>) +} + +impl Expression { + fn interpret(&self, tokens: &[Token]) -> bool { + match self { + Self::Literal(test) => for token in tokens { + if token.value() == test { + return true; + } + + false + }, + Self::Or(expr1, expr2) => expr1.interpret(tokens) || expr2.interpret(tokens), + Self::And(expr1, expr2) => expr1.interpret(tokens) && expr2.interpret(tokens), + } + } +} +</pre> + +<p> + Let's say, for the sake of argument, that you want the set of expressions to be extendable. That can + be done too. +</p> + +<pre> +pub trait Expression { + fn interpret(&self, tokens: &[Token]) -> bool; +} + +pub struct TerminalExpression { + literal: Arc<str>, +} + +impl TerminalExpression { + pub fn new(literal: Arc<str>) -> Self { + Self { literal } + } +} + +impl Expression for TerminalExpression { + fn interpret(&self, tokens: &[Token]) -> bool { + for token in tokens { + if token.value() == self.literal { + return true; + } + } + + false + } +} + +pub struct OrExpression { + expression1: Arc<Expression>, + expression2: Arc<Expression>, +} + +impl OrExpression { + pub fn new(expression1: Arc<Expression>, expression2: Arc<Expression>) -> Self { + Self { + expression1, + expression2 + } + } +} + +impl Expression for OrExpression { + fn interpret(&self, tokens: &[Token]) -> bool { + self.expression1.interpret(tokens) || self.expression2.interpret(tokens) + } +} + +pub struct AndExpression { + expression1: Arc<Expression>, + expression2: Arc<Expression>, +} + +impl AndExpression { + pub fn new(expression1: Arc<Expression>, expression2: Arc<Expression>) -> Self { + Self { + expression1, + expression2 + } + } +} + +impl Expression for AndExpression { + fn interpret(&self, tokens: &[Token]) -> bool { + self.expression1.interpret(tokens) && self.expression2.interpret(tokens) + } +} +</pre> + +<p>But that is far more verbose.</p> + +<h3>Observer</h3> + +<p> + In case someone is now accusing me of picking examples that work too easily in Rust, the Observer + pattern is actually quite difficult. So before I show this one, I want to give a sneak peak into what + this pattern looks like in Java. The inheritance version is provided on + <a href="https://www.oodesign.com/observer-pattern">oodesign.com</a>, but I'll use composition. +</p> + +<pre> +interface Observer { + void update(); +} + +class Observable { + private ArrayList<Observer> observers; + + public Observable() { + this.observers = new ArrayList<>(); + } + + public void attach(Observer observer) { + this.observers.add(observer); + } + + public void detach(Observer observer) { + this.observers.remove(observer); + } + + public String notify(String newState) { + for (Observer observer in this.observers) { + observer.update(); + } + } +} + +class StatefulObservable { + private Object state; + private Observable observable; + + public StatefulObservable(Object initialState) { + this.state = initialState; + this.observable = new Observable(); + } + + public void attach(Observer observer) { + this.observable.attach(observer); + } + + public void detach(Observer observer) { + this.observable.detach(observer); + } + + public Object getState() { + return this.state; + } + + public setState(Object newState) { + this.state = newState; + this.observable.notify(); + } +} +</pre> + +<p> + I wanted this here because I wanted to point out that the tricky part of this pattern is Rust's + borrow checker, and not composition. So now we'll walk step by step through making this work in Rust. +</p> + +<p>First, we'll need an <code>Observer</code> trait. That's easy.</p> + +<pre> +trait Observer { + fn update(&mut self); +} +</pre> + +<p> + "Show me your data, and your functions will be obvious", so let's go ahead and make the + <code>Observable</code> struct. The problem is that we need to be able to mix and match different + types that all implement <code>Observable</code>. We'll need a <code>dyn Observable</code>, but + trait objects aren't sized, and thus can't be put into a <code>Vec</code> directly. We can't use + <code>Arc</code> because <code>Observer::update</code> requires a mutable reference. We'll use + <code>Box</code>. +</p> + +<pre> +struct Observable { + observers: Vec<Box<dyn Observer>>, +} +</pre> + +<p> + You might be wondering why <code>Observer::update</code> needs a mutable reference. What if we want + to have multiple references to the observer? The answer to the first question is that some + <code>Observer</code>s will need to mutate after an update. The answer to the second question is that + it won't be difficult to wrap your existing <code>Observer</code> inside of an <code>Arc</code> + before attaching it to the <code>Observable</code>. Yeah, <code>Box<Arc<Observer>></code> + isn't ideal, but it's better than forcing the Observables to all have interior mutability. +</p> + +<pre> +use super::ImmutableObserver; // we want multiple references to this + +#[derive(Debug, Clone)] +pub struct AliasableObserver(Arc<ImmutableObserver>); + +impl Observer for AliasableObserver { + fn update(&mut self) { + self.0.update() + } +} +</pre> + +<p>Alternatively, we could update our API to allow both immutable and mutable observers.</p> + +<pre> +pub trait Observer { + fn update(&self); +} + +pub trait MutableObserver { + fn update(&mut self); +} + +pub struct Observable { + observers: Vec<Arc<dyn Observer>>, + mutable_observers: Vec<Box<dyn MutableObserver>>, +} +</pre> + +<p> + I actually like this idea, so let's go with this. But this raises another concern. In Java, we could + detach observers by passing in a reference and comparing the pointer addresses. This is still + possible with <code>observers</code> but not <code>mutable_observers</code>, since a <code>Box</code> + needs to be a unique reference. But, we could use a raw pointer. +</p> + +<pre> +impl Observable { + pub fn detach(&mut self, observer: *const ()) -> Option<()> { + for i in 0..self.observers.len() { + if std::ptr::addr_eq(&self.observers[i], observer) { + self.observers.remove(i); + return Some(()); + } + } + + for i in 0..self.mutable_observers.len() { + if std::ptr::addr_eq(&self.mutable_observers[i], observer) { + self.mutable_observers.remove(i); + return Some(()); + } + } + + None + } +} +</pre> + +<p> + Dealing with raw pointers is annoying, but it works. Alternatively, we could put an ID on each + attached observer, and remove using an ID, but that sounds even worse. Plus, we'd have to deal with a + sparse index. I don't want that. +</p> + +<p> + I suppose we've jumped the gun a bit on implementing <code>detach</code>, so now let's implement the + rest of <code>Observable</code>. +</p> + +<pre> +impl Observable { + pub const fn new() -> Self { + Self { + observers: Vec::new(), + mutable_observers: Vec::new(), + } + } + + pub fn attach(&mut self, observer: Arc<dyn Observer>) { + self.observers.push(observer); + } + + + pub fn attach_mutable(&mut self, observer: Box<dyn MutableObserver>) { + self.mutable_observers.push(observer); + } + + pub fn detach(&mut self, observer: *const ()) -> Option<()> { + // ... already did this ... + } + + pub fn notify(&mut self) { + for observer in &self.observers { + observer.update(); + } + + for observer in &mut self.mutable_observers { + observer.update(); + } + } +} +</pre> + +<p> + Now all that's left is the <code>StatefulObserver</code>. In the Java implementation, I used an + <code>Object</code> as the state, but that's not very useful in Rust (or in general). So instead + I'll use a generic type here. The implementation is pretty simple, so I'll write that out too. +</p> + +<pre> +pub struct StatefulObservable<T> { + state: T, + observable: Observable, +} + +impl<T> StatefulObservable<T> { + pub const fn new(initial_state: T) -> Self { + Self { + state: initial_state, + observable: Observable::new(), + } + } + + pub fn state(&self) -> &T { + &self.state + } + + pub fn attach(&mut self, observer: Arc<dyn Observer>) { + self.observable.attach(observer) + } + + pub fn attach_mutable(&mut self, observer: Box<dyn MutableObserver>) { + self.observable.attach_mutable(observer) + } + + pub fn detach(&mut self, observer: *const ()) -> Option<()> { + self.observable.detach(observer) + } + + pub fn set_state(&mut self, new_state: T) { + self.state = new_state; + self.observable.notify(); + } +} +</pre> + +<p>And now we're done. Here's the full code for the curious:</p> + +<pre> +use std::sync::Arc; + +pub trait Observer { + fn update(&self); +} + +pub trait MutableObserver { + fn update(&mut self); +} + +pub struct Observable { + observers: Vec<Arc<dyn Observer>>, + mutable_observers: Vec<Box<dyn MutableObserver>>, +} + +impl Observable { + pub const fn new() -> Self { + Self { + observers: Vec::new(), + mutable_observers: Vec::new(), + } + } + + pub fn attach(&mut self, observer: Arc<dyn Observer>) { + self.observers.push(observer); + } + + + pub fn attach_mutable(&mut self, observer: Box<dyn MutableObserver>) { + self.mutable_observers.push(observer); + } + + pub fn detach(&mut self, observer: *const ()) -> Option<()> { + for i in 0..self.observers.len() { + if std::ptr::addr_eq(&self.observers[i], observer) { + self.observers.remove(i); + return Some(()); + } + } + + for i in 0..self.mutable_observers.len() { + if std::ptr::addr_eq(&self.mutable_observers[i], observer) { + self.mutable_observers.remove(i); + return Some(()); + } + } + + None + } + + pub fn notify(&mut self) { + for observer in &self.observers { + observer.update(); + } + + for observer in &mut self.mutable_observers { + observer.update(); + } + } +} + +pub struct StatefulObservable<T> { + state: T, + observable: Observable, +} + +impl<T> StatefulObservable<T> { + pub const fn new(initial_state: T) -> Self { + Self { + state: initial_state, + observable: Observable::new(), + } + } + + pub fn state(&self) -> &T { + &self.state + } + + pub fn attach(&mut self, observer: Arc<dyn Observer>) { + self.observable.attach(observer) + } + + pub fn attach_mutable(&mut self, observer: Box<dyn MutableObserver>) { + self.observable.attach_mutable(observer) + } + + pub fn detach(&mut self, observer: *const ()) -> Option<()> { + self.observable.detach(observer) + } + + pub fn set_state(&mut self, new_state: T) { + self.state = new_state; + self.observable.notify(); + } +} +</pre> + +<h2>Conclusion</h2> + +<p> + Anything you can do with inheritance, you can also do with composition. Composition will make your + code easier to modify in the future, without creating weird semantics or bugs. Use composition, + wherever possible. +</p> + +</article> + +<hr/> + +<footer> + <ol> + <li id="footnote-1"> + I'm not quite sure what software engineering majors do at RIT. I imagine them all sitting + around a table, and saying words like "agile" and "scrum" with no context. I took Computer + Science. <a class="return" href="#af-1">return</a> + </li> + </ol> +</footer> + +</body> diff --git a/src/blog/best-tool-for-the-job.kuht b/src/blog/best-tool-for-the-job.kuht new file mode 100755 index 0000000..ad89290 --- /dev/null +++ b/src/blog/best-tool-for-the-job.kuht @@ -0,0 +1,685 @@ +<import "base.kuht" as "base" /> + +<head> + <title>The Best Tool for the Job</title> + <meta name="description" content="What language should you use for different tasks? Don't follow my advice" /> +</head> + +<body> + +<article> + +<h1>The Best Tool for the Job</h1> + +<p> + I often see programmers saying that programming languages are tools, and that instead of arguing over + what the best language is, we should just pick the right tool for the job. I think this is a sensible + philosophy, but I've never seen it used in a sensible way. One time I was talking to someone who told + me that the Software Engineering program at RIT exclusively uses the Java programming language. When + I pointed out how weird this was, he said, "Languages are tools, and you should just use the right + tool for the job"<a id="af-1" href="#footnote-1"><sup>1</sup></a>. If Java is a hammer in this + analogy, then why is this major not teaching you how to use a screwdriver? +</p> + +<p> + There are other times when the philosphy is used in sort of the right way, but to promote a language + that doesn't fit at all. I keep seeing Go being recommended for web development, for example. That is + the intended purpose of the language, but I don't recommend it due to the strange way it handles + errors<a id="af-2" href="#footnote-2"><sup>2</sup></a>. +</p> + +<p> + To be clear, I don't hate this way of thinking. I approve of it. In fact, I find the suggestion that + we should just use the same language for everything even worse. But I wish more thought went into the + idea. In this article, I want to propose the actual ideal language for each task. Surely, this will + end all debates related to the choice of programming language. If you currently in the middle of such + a debate, link this article to your co-workers<a id="af-3" href="#footnote-3"><sup>3</sup></a>, and + give yourselves the rest of the day off. +</p> + +<h2>My Biases</h2> + +<p> + Just like everybody else, I have my own preferences for language design, which might not align with + yours. I still think my recommendations are correct, but I'll make some premises clear here, so that + if you want to adapt my work for your preferences, you may do so. +</p> + +<ul> + <li> + I hate dynamic typing. The time it takes me to ensure type safety is far less than the time it + takes me to debug type errors, find the documentation that tells me what types I can pass into + a function, figure out what methods I can call on a given type, or even what type a given value + is supposed to be. Ensuring type safety doesn't take that much time anyway, as long as your + language has type inference. Most of the languages I recommend here have type inference. + </li> + <li> + On that note, the language should try to prevent me from making runtime errors. No language can + fully prevent logic errors, but you shoud note that most, if not all, of the languages I + recommend have some sort of null safety, albeit not always by default. + </li> + <li> + Similarly to the last two points, borrow checking provides many benefits (including + <a href="/blog/how-happylock-works.html">deadlock prevention</a>) which I believe are worth the + hassle. I don't recommend a borrow checker all of the time, but it is a pretty common occurrence. + </li> + <li> + I don't like object-oriented programming. I don't have a problem with classes. But when a + language forces me to put all of my behavior inside of a class, I get annoyed. I also recommend + against using inheritance. I can put up with languages like C#, but it's not an ideal situation. + </li> +</ul> + +<p> + Something else that I think is worth pointing out is that I will be focusing more on the language + features than the ecosystem. In a more practical discussion, you should probably be thinking about + this. But I think the language features are more interesting, and it's not hard to imagine a good + ecosystem being developed for any of these languages. Sometimes, a language feature means it might + be easier to develop an ecosystem designed for a specific task. +</p> + +<h2>Systems or Embedded Programming</h2> + +<p> + You'll need Assembly. There's no getting around that. The question is, what language do you use to + call your Assembly utilities? Given that most programming languages need a runtime, and that the + runtime does not exist until you write it, you have a limited set of options: +</p> + +<ul> + <li><strong>More Assembly:</strong> Yuck. Do I need to explain this one?</li> + <li><strong>C:</strong> + An oldie. Not quite a goodie. I think we can do better than 30 years of tech debt. + </li> + <li><strong>C++:</strong> + Not much better than C. Linus Torvalds would actually prefer C. There's a lot of language bloat, + and creating a custom standard library for C++ is much harder than for C. + </li> + <li><strong>Rust:</strong> + Not exactly made for kernel development. The Linux kernel can use it now, but they modified some + of the standard library features in order to make it work. + </li> + <li><strong>Zig:</strong> This language isn't even 1.0 yet.</li> +</ul> + +<p> + I recommend <strong>Rust</strong>. Most of the errors in many projects are due to memory safety issues. + <a href="https://security.googleblog.com/2022/12/memory-safe-languages-in-android-13.html"> + Android has solved this problem by using more memory safe languages</a>. + Windows and Linux are also starting to introduce Rust into their codebases for the same reason. +</p> + +<p> + I also mentioned having to write a runtime for C and C++. That's not necessary for Rust. Rust has, in + addition to the standard library, a core library. The core library does not require an operating + system, and has some very convenient utilities, + <a href="https://doc.rust-lang.org/stable/std/primitive.str.html#method.encode_utf16">including UTF-16</a>. + And even without the standard library, you can use the built-in alloc crate to use utilities that require + memory allocation. You'll just need to bring your own memory allocator. +</p> + +<p> + I'm not done yet. Rust has async/await. Rust's implementation of async/await does not rely upon the + existence of an operating system. You can create your own async runtime, and then use async/await + inside of your kernel! If that does not sell you, nothing will. +</p> + +<h2>Parsers and Compilers</h2> + +<p> + Now you have an operating system. Maybe you already have a shell and some command-line utilities. But + you now you need a way to compile programs on your new system. You don't necessarily want to write + everything in Rust. But you'll need to write a compiler for that to work. What language do you do it + in? I have a couple ideas for a good parser language. I'll make the case for both. +</p> + +<h3>Rust</h3> + +<p> + There's one particular reason why I like Rust here: enums. Rust enums are discriminant unions, which + can make it much easier to create a token type, and an abstract syntax tree. Here's an example of + some types I'd make when creating a basic Lisp interpreter: +</p> + +<pre> +enum TokenType { + LeftParenthesis, + RightParenthesis, + + LiteralNumber(f64), + LiteralString(Arc<str>), + Symbol(Arc<str>), +} +</pre> + +<p>And here's what the abstract syntax tree could look like:</p> + +<pre> +enum ExpressionType { + LiteralNumber(f64), + LiteralString(Arc<str>), + Symbol(Arc<str>), + + SExpression(Arc<[ExpressionType]>), +} +</pre> + +<p> + It gets more complicated than that when you're doing error reporting, but you can at least understand + how this makes things easier. I also recommend you take a look at the + <a href="https://en.wikipedia.org/wiki/Icon_(programming_language)">Icon programming language</a>. + You won't be using it, but it is a very useful framework for tokenizing. I wrote a Rust + implementation of some of Icon's standard library, in the form of a library called + <a href="https://www.lib.rs/snob">snob</a>. I still use it, and I recommend giving it a try. +</p> + +<p> + I don't think Rust is the perfect language for this task though. You may have noticed all of my + <code>Arc</code>s in that code snippet. Constantly copying a <code>Box</code> or a <code>Vec</code> + will get expensive very quickly. But this interface isn't the greatest. Constantly copying an + <code>Arc</code> isn't great either. In fact, the D programming language found that typical + generational garbage collection is faster than implementing garbage collection using + <code>Arc</code><a id="af-4" href="#footnote-4"><sup>4</sup></a>. Another problem is that Rust uses + more memory than a garbage collected language would. Rust monomorphizes generic functions for more + speed. But this bloats the binary size and uses a lot of memory. I can't find it now, but I remember + seeing a GitHub issue stating that rustc couldn't be compiled on a Raspberry Pi, because compiling it + requires too much memory. +</p> + +<p> + So I would recommend a garbage collected language, but the problem is that I can't find a language + that has Rust-style enums and garbage collection<a id="af-5" href="#footnote-5"><sup>5</sup></a>. C# + could do this once they figure out <a href="https://github.com/dotnet/csharplang/issues/113">discriminated unions</a>. + But I don't hold much hope for that right now. +</p> + +<h3>Dart</h3> + +<p> + Now I'll admit that I've never attempted to write a parser in Dart before. I'm only thinking about it + now. Dart doesn't have Rust-style enums. But it tried. Dart has sealed classes, and if you run a + switch statement on a sealed class, then it will require an exhaustive match. Here, I'll duplicate + the Token data structure I showed in Rust: +</p> + +<pre> +sealed class Token {} + +class LeftParenthesis extends Token {} + +class RightParenthesis extends Token {} + +class LiteralNumber extends Token { + final double value; + + const LiteralNumber(this.value); +} + +class LiteralString extends Token { + final String value; + + const LiteralString(this.value); +} + +class Symbol extends Token { + final String name; + + const Symbol(this.name); +} +</pre> + +<p> + Notice how that was far more verbose than the Rust implementation. And I didn't even create an + abstract syntax tree. We're also using inheritance, so we'll have to be careful about how we modify + the base class. But there is one benefit. Something I did not show in the Rust implementation is the + information about where the data is stored in the original source code. Adding that to the Rust + implementation can be done using another struct. +</p> + +<pre> +struct Token { + token_type: TokenType, + start: usize, + end: usize, + file: Arc<Path>, +} +</pre> + +<p> + Then any token we create has to be wrapped in this new structure. But in Dart, we can do it by + modifying the base class +</p> + +<pre> +sealed class Token { + final int start; + final int end; + final File file; + + const Token({required this.start, required this.end, required this.file}); +} +</pre> + +<p> + But then we'll also have to modify the constructors on the child classes to have the same + information. So you might still want to use composition here. But this would cut down on the of times + you have to write something like <code>token.token_type</code>. +</p> + +<p>Dart seems promising. I'll try writing a parser in it sometime.</p> + +<h2>Scripts</h2> + +<p> + So now you can create a parser. But a parser for what language? Say you need a scripting language for + some simple tasks. What can you use? +</p> + +<p> + Bash is ok, if the only thing you need to do is run some simple commands. But most scripts will end + up being more complicated than that. If your bash script needs a variable, then you shouldn't use + Bash. +</p> + +<p> + Here we have a different set of priorities. This code shouldn't take too long to write. In many cases + it will be thrown away immediately. We should still make it easy to read, just in case. But we might + not need it for very long. +</p> + +<p> + <strong>Python</strong> was designed to be a middle-ground between Bash and C. It's painfully slow, + but our scripts shouldn't take too long to run anyway. The syntax is kinda annoying, but not annoying + enough that I would consider writing a new language to replace it. There are some conveniences + like default parameters that I honestly wish Rust had, but that one in particular was implemented + with some stupid mutability rules. There's no interfaces, only inheritance. + <a href="https://www.youtube.com/watch?v=sbVxq7nNtgo">Exceptions are bad</a>. After getting used + to the borrow checker, I don't appreciate the <code>with</code> blocks. So much of the standard library + is deprecated. Python is certainly not the perfect language. +</p> + +<p> + But it does come with some interesting ideas. Gradual typing was probably the right decision for a + language like this. And I can't complain, because the type system comes with null safety. The idea of + having big integers by default is also really clever. And the large set of builtins is both a + blessing and a curse. It's a good idea for scripting, but not when half of the builtins are + deprecated. The package manager installs everything globally, which is not a great idea, but nothing + else really makes sense for a scripting language. Most scripts don't need their own directory and + <code>node_modules</code>. +</p> + +<p> + That said, there's more that could be done. I don't like whitespace as syntax, but maybe we could + agree on automatic-semicolon-insertion as a middle ground. A derive-like syntax seems especially + important for a language trying to cut down on developer time. This could even be done with macros. + Speaking of which, I'd like to see interfaces, and Rust's enums. Dart has some nice ideas too, like + the <code>this.x</code> parameter in constructors, as well as named constructors and factories. Hell, + use 1-based-indexing and use a numerical tower as the default number type, or at least use a decimal + type. Python is used mostly by novices, so we might as well make it easier for them. +</p> + +<p> + At this point, I've essentially proposed my own programming language. To be clear, I don't hate + Python enough to actually make this. I'm just dreaming here. +</p> + +<h2>Game Development</h2> + +<p> + Now you want to have some fun, so why not make a video game? What language can we use for that? +</p> + +<p> + There are several constraints that game engines have that other programs do not. For example, we + really do not want any stuttering, so a garbage collector is out of the question, unless you decide + to manually run the garbage collector on every single frame. However, performance is another + important goal for a video game, so we probably want to avoid that too. That means + <strong>Rust</strong> is our only option. +</p> + +<p> + The problem there is that the ecosystem for Rust game development + <a href="https://www.arewegameyet.rs">isn't very ripe</a>. There is the work-in-progress + <a href="https://bevyengine.org/">Bevy</a> game engine, but I honestly don't like the way you query + for components in it. So you'll be making your own game engine. Given that this is Rust, we will + still use an entity-component system. But the queries will be much simpler: +</p> + +<pre> +let mut player_sprite = cx.sprites.get("player")?; +let player_collider = cx.colliders.get("player")?; +player_sprite.set_position(player_collider.position()); +</pre> + +<p> + It would be really cool if the engine supported hot reloading scripts, but Rust doesn't have that. + We could make it work by compiling the scripts to WebAssembly and reloading them when the file + changes. But this would introduce undefined behavior if the old script's state doesn't match what the + new script is expecting (i.e. a struct definition changed). To resolve this, we will require that + scripts not store any state. Instead, the memory of the script will be cleared every frame. Any + information we need to keep will be stored in the engine code. +</p> + +<pre> +fn run(state: ScriptState) { + let mut velocity: i32 = state.get("player_velocity").parse(); + velocity += 1; + state.store("player_velocity", velocity); +} +</pre> + +<p> + Now, something magical has just happened. We are clearing all state every single frame. All + heap-allocated memory, as far as this script is concerned, is static. This will be the new + memory-allocation strategy. We can use a bump allocator as the allocator. And that means we won't + need a garbage collector (because we're clearing memory every frame), or a borrow checker + (because all heap memory is static). It wouldn't be too hard to adapt an existing garbage-collected + language to use this strategy, so we could just pick one that can theoretically be compiled to + WebAssembly, and make a compiler that doesn't insert a garbage collector. That could include C#, + Dart, Kotlin, or Swift. But we could also make a custom language suited to this task. This could + make it even faster than Rust, because bump allocators are so fast. +</p> + +<p> + In fact, I don't think the perfect language for game development exists yet. I want something easy to + embed, like Lua, but statically typed<a id="af-6" href="#footnote-6"><sup>6</sup></a>. If we could + block the use of the file system or network, then we could ensure that mods don't contain any + viruses. But the language server would need some way to know what libraries are being imported, and + what they contain. +</p> + +<p>Another useful idea is a property declaration. Here's an analogy:</p> + +<pre> +// this is determined at compile-time, and not modifiable at runtime +const PLAYER_SPEED: usize = 5; + +// this can be mutated (not in Rust, but in other languages), but an initial value must be given +static PLAYER_SPEED: usize = 5; + +// this is a value inserted by the game engine +property PLAYER_SPEED: usize; +</pre> + +<p> + The engine would need to parse the script to determine what properties exist, give the developer a + slider that they can use to change it, and then insert that value into the script. Technically, this + would probably just be syntactic sugar for calling a function. But you wouldn't have to worry about + typos in the property name. +</p> + +<p> + I'm sure are other features that I think could be useful as well, but this proves the point that a + language specific to game scripting could be useful. +</p> + +<h2>Backend Web</h2> + +<p> + Now you have a video game. Maybe this game has online features. In that case, you'll need a web + server to handle communication between players. Once again, we need to pick a language to write it + in. Let's take a look at the usual suspects. +</p> + +<h3>Rust (or Swift)</h3> + +<p> + Rust is great because of the exceptional error handling. If all errors are handled using the + <code>Result</code> type, and errors are sent to the client using the <code>Result</code> type, + then you guarantee no cryptic 500 errors, except in the case of a panic. Swift also has a Result + type, so most of what I say here should apply to Swift as well (although I'm not familiar enough + with Swift to tell you when this advice might not apply). +</p> + +<p> + A point to Rust specifically is the performance and lack of a garbage collector. In most languages, + when the garbage collector starts, the entire program stops. But Rust doesn't have this problem, so + you can send responses at the speed of the network. Another point to Rust is the fearless + concurrency. No data races ever, and bad race conditions should be rare as well. Multithreading is + important to a server, so this is a good thing to have. The ecosystem is also able to provide things + that no other language can. <a href="https://www.lib.rs/serde">Serde</a> makes JSON and XML + serialization/deserialization a piece of cake. Take a look at <a href="https://lib.rs/sqlx">SQLx</a>. + Not only can Rust get rid of memory and race errors, it also gets rid of SQL errors. +</p> + +<p> + AWS supports Lambda functions written in Rust, and even gives you a discount if you do so, since it + runs faster. If you don't like cold boots, then you can always use an actual server application. + There are a few backend web frameworks written in Rust, such as <a href="https://www.lib.rs/axum">axum</a>, + which everyone seems to love despite the fact that it isn't finished yet. I use + <a href="https://www.lib.rs/actix-web">actix-web</a>. It's slightly faster than axum, and it's both + stable and production-ready. +</p> + +<h3>Dart</h3> + +<p> + Let's start with what Dart doesn't have. It doesn't have anything as nice as serde or SQLx because it + doesn't have macros. It also doesn't have a <code>Result</code> type. AWS Lambda support for Dart is + also lacking. So all in all, I recommend <strong>Rust</strong> over Dart. But I wanted to mention + Dart because of the interesting strategy it uses to avoid data races. +</p> + +<p> + Dart doesn't allow you to share memory between threads. Each thread is its own <code>Instance</code> + with its own heap. But the <code>Instance</code>s can communicate with each other over channels. So + there are no data races, just like Rust. +</p> + +<p> + Now, Dart does have a garbage collector. It's always going to be slightly slower than Rust under + load. But since all of the threads have a separate heap, the garbage collector does not need to stop + your application from running. It only stops one thread. And the garbage collector can be programmed + to run early if a thread has no tasks waiting for it. So in practice, you might never experience a + GC pause. +</p> + +<p> + There are a number of Dart backend frameworks. I haven't personally tried any of them, and many of + them are deprecated, but I think it's worth checking out if you really want to avoid garbage + collection. Some of them have built-in ORMs and I've seen at least one with built-in OAuth2. +</p> + +<h3>WebAssembly</h3> + +<p> + In case you didn't guess, we're going down the same path I followed in my game development ideas. +</p> + +<p> + The framework itself should be written in Rust, but we can do better for the route code. I really + like the idea of serverless, but cold starts are a problem for me. But take a look at Fermyon's + <a href="https://www.fermyon.com/spin">spin</a>. It uses WebAssembly to get the best of both worlds. + Just like serverless functions, they don't need to be loaded up all of the time. But because + WebAssembly is already heavily sandboxed, they don't need to create a virtual machine to contain your + code. Which makes the cold starts fast. So fast that they've decided to just cold start every single + time. With the help of WASI, Spin can support other languages, such as Go, Python, JavaScript, C#, + and Ruby. +</p> + +<p> + So you might see where this is going. Again, we can use a bump allocator to avoid garbage collection + and borrow checking. I think spin uses a custom implementation of WASI in order to keep everything + sandboxed, so we could do that too. But, again, custom compilers would be needed, so we could forego + WASI and use a more stable interface that we invent. +</p> + +<p> + I also think that a new language to support this system could be helpful. We'll need Rust macros, + Rust enums, and a <code>Result</code> type. Although, the macros would probably be implemented in the + embedding, in order to make it more powerful. We'll also want a multithreading system similar to the + one in Dart. Of course this new language would need async/await. Ideally, it would also have a core + library separate from the main one. Ideally, it would also be impossible to panic, but that seems + impossible. +</p> + +<p> + There's something else we can do now that we're running everything inside of a WebAssembly framework. + Spin provides its own key/value store. The neat thing about this is that we don't necessarily need a + network request to access this store. It's stored locally on the machine. So we can do fast DB access + by creating our own database, and embedding it directly into the framework. I think Spin had this + idea with the built-in SQLite database, but + <a href="https://www.sqlite.org/whentouse.html#situations_where_a_client_server_rdbms_may_work_better"> + SQLite doesn't work well in some cases + </a>. + The key/value store Spin provides is also limited to about 1000 tuples, if I understand correctly. + So I would prefer something with a built-in wide column DB, that doesn't require a network connection + to access. Maybe also a relational database, but those are harder to create from scratch (speaking + from experience). +</p> + +<p> + I'm not actually convinced that WebAssembly is the best possible thing here either. WebAssembly is a + stack-based bytecode, which is typically slower than one based on registers. This is a good idea for + WebAssembly, because it's on the web, and stack-based bytecode is much smaller than register-based. + So you'll be able to download the code more quickly. But a server doesn't have this requirement. The + bytecode will need to be uploaded to the server, but the size doesn't matter too much after that. So + I could imagine a new bytecode being invented here that takes advantage of registers. It could even + be AOT-compiled once it reaches the server. The same idea applies to games as well. The only benefit + to a smaller bytecode would be a smaller download size, but performance is probably more important + there. +</p> + +<h2>Frontend Web</h2> + +<p> + Any language you choose here is going to either compile into JavaScript or WebAssembly. WebAssembly + can be a problem for many languages, since you need glue code to make the language's runtime work + in a browser, which is often the same size as a JavaScript framework. If there were a language that + had a runtime more directly based on JavaScript, designed specifically for frontend WebAssembly, that + could work<a id="af-7" href="#footnote-7"><sup>7</sup></a>. But until that happens, I'm going to + recommend avoiding this approach. +</p> + +<p> + That leaves your options to JavaScript, and languages that compile to JavaScript. This shouldn't be + a surprise to anyone, but <strong>TypeScript</strong> is your best bet here. I don't think it's + perfect. It takes some of the flaws of JavaScript for itself. But it does make many of them more + bearable. It doesn't cause the JavaScript to get much larger. You get built in lowering to older + versions of JavaScript. And you inherit the whole JavaScript ecosystem. +</p> + +<p> + Alright, here's the part where I propose a better language. First of all, we shouldn't be inheriting + JavaScript's syntax. If we have to compile to JavaScript from another language, why not compile from + a good one? So things like <code>===</code> don't need to be there. We should also get rid of + <code>undefined</code>. +</p> + +<p> + Now, one problem I've noticed in most JavaScript frameworks is the difficulty in figuring out which + variables have been updated. Svelte tried to solve this by using the equal sign as a marker that a + variable needs to be updated. But what if I use <code>array.push(5)</code>? Now I need to enter a + hacky <code>array = array</code> to make the change visible. So we'll want some way of detecting + which methods mutate a value. +</p> + +<p> + Rust, as you may know, has that. Methods that mutate the value take <code>&mut self</code>. How would + we implement this system in Rust? +</p> + +<pre> +struct Signal<T> { + value: T, +} + +struct SignalGuard<'a, T> { + signal: &'a mut Signal<T>, +} + +impl<'a, T> Drop for SignalGuard<'a, T> { + fn drop(&mut self) { + mark_as_updated(self.signal); + } +} +</pre> + +<p> + We won't be able to replicate this in TypeScript because TypeScript doesn't have a borrow checker, + and thus cannot detect when a value needs to be dropped. That's the problem that the garbage + collector is meant to solve. So how about we add a borrow checker to TypeScript? +</p> + +<p> + Yes, I'm actually proposing taking a language which does not have or need a borrow checker, and + adding a borrow checker to it. The borrow checker just makes so many problems easier. For instance, + we said that we needed a language that compiles to WebAssembly but is more closely related to + JavaScript. If we have a borrow checker, and plan the language properly, this language could even + compile to WebAssembly, making it faster than JavaScript. +</p> + +<p> + I've been working on a web framework for a bit now, and I think this would be a useful feature. I + think I'll try it someday. +</p> + +<h2>Conclusion</h2> + +<p> + I didn't mention desktop or mobile apps, but the answer there would be incredibly simple: Dart. Dart + was designed with that specifically in mind, and went through extensive user testing to ensure that + it could be useful for that task. I wouldn't even have any suggestions for what to add to it + (other than Rust enums and macros). Other languages try to have more general purposes, but if a + language is built with a specific goal in mind, it will be good for that task. Python was designed + with scripts in mind, and I recommended it, even though I hate dynamically typed languages. Rust was + designed as a competitor to C++, and it does indeed replace C++ in many use-cases. Icon is a language + that barely exists, and yet I still gave it a shoutout in the parser section. +</p> + +<p> + The idea of languages being tools is a good one. But I don't think enough thought goes into it. Make + a language with a task in mind. It will be a great one. +</p> + +</article> + +<hr /> + +<footer> + <ol> + <li id="footnote-1"> + This guy is a jerk, for the record. This is not an important detail for the story. But I did + want the world to know that. <a class="return" href="#af-1">return</a> + </li> + <li id="footnote-2"> + I recommend reading some of Amos' articles on Go. The best one is + <a href="https://fasterthanli.me/articles/abstracting-away-correctness">Abstracting away correctness</a>, + but there's <a href="https://fasterthanli.me/tags/golang">other good ones too</a>. + <a class="return" href="#af-2">return</a> + </li> + <li id="footnote-3"> + If you want to be sneaky, read ahead and make sure I'm confirming your biases before linking + here. Either way, you're boosting engagement, so I don't mind. If you're a Rust diehard, I'll + save you the time and tell you that I recommend Rust for most things. As you read, you'll + probably notice that following my advice might not be in your company's best interest. + <a class="return" href="#af-3">return</a> + </li> + <li id="footnote-4"> + This criticism doesn't apply to most Rust codebases, because most codebases don't use + <code>Arc</code> on every single variable. <a class="return" href="#af-4">return</a> + </li> + <li id="footnote-5"> + You might be wondering why I didn't mention functional languages here. The main reason is + that I don't think I'm familiar enough with them to give a good answer. My impression is that + they are good for parsing, but not as good for interpreting or generating machine code. + <a class="return" href="#af-5">return</a> + </li> + <li id="footnote-6"> + Dart is an embedded scripting language, and statically typed, but it's not easy to embed. + There's zero documentation on how to do that, so you just have to rely on example code. But + you're gonna have to write your own compiler anyway, so it doesn't really matter that the + reference implementation is hard to embed. The hard part will be rewriting the Dart compiler. + <a class="return" href="#af-6">return</a> + </li> + <li id="footnote-7"> + I hear a voice in my head screaming <strong><em>ASSEMBLYSCRIPT!!</em></strong> If I'm being + honest, AssemblyScript seems like a good language. The reasons I have for not recommending it + are that it's not 1.0, and + <a href="https://www.assemblyscript.org/standards-objections.html">the developers give me bad vibes</a>. + <a class="return" href="#af-7">return</a> + </li> + </ol> +</footer> + +</body> diff --git a/src/blog/color-spaces.kuht b/src/blog/color-spaces.kuht new file mode 100644 index 0000000..ee668b5 --- /dev/null +++ b/src/blog/color-spaces.kuht @@ -0,0 +1,130 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Color Spaces</title> + <meta name="description" content="Colors are more interesting than you may think" /> +</head> + +<body> +<article> +<h1>Color Spaces</h1> + +<p> + Colors are more interesting than you might think. I want to share a bit of + information on colors so that you can pick better ones. +</p> + +<h2>RGB</h2> + +<p> + Most programmers should already be familiar with RGB. When you have a hex code + like <code>#e395fa</code>, that's an RGB value. Displays are made of pixels + which can simultaneously display red, green, and blue lights. The hex code + tells us how much of each color to use, using a two-digit hexadecimal number. + This format is common because it maps directly to what is displayed on the + screen, so most other color spaces need to be converted to RGB eventually. +</p> + +<h2>HSL</h2> + +<p> + RGB, despite its necessity, is pretty hard to intuitively understand. I gave + the example of <code>#e395fa</code> earlier, but it's difficult to imagine what that color + actually looks like (it's lavender-ish). HSL is a 1-1 mapping of RGB, meaning + that each HSL value maps to a different RGB value. And it's in a form that is + pretty intutive. H stands for hue, and is measured in degrees. It points to a + specific color on the color wheel. 0° is red, 180° is turquoise, 270° is + purple, and so on. S stands for saturation, which refers to the strength of the + color. Lower saturations mean that the color will be less colorful, and look + more gray-ish. L stands for lightness. Whiter colors will have higher + lightness, and blacker colors will have lower lightness. +</p> + +<h2>LAB</h2> + +The problem with HSL is that is is not perceptually uniform. According to HSL, all of these colors are the same lightness. + +<img + alt="The HSL color space at 60% lightness. Some parts are clearly lighter than others" + width="360" + height="200" + src="https://www.hsluv.org/images/hsl.png" +/> + +<p> + LAB, otherwise known as CIELAB, was designed by the International Commission + of Illumination to be more perceptually uniform. L, once again, stands for + lightness, and is on a scale of zero to one. A and B don't stand for anything, + don't have any bounds, and can in fact be negative. A is the spectrum between + red and green, and B is the spectrum between blue and yellow. +</p> + +<p> + LAB still isn't perfect. The most notable problem is known as hue shift. As the + color gets lighter, the color starts somehow shifting towards purple. +</p> + +<img + alt="Demonstration of the hue shift described above" + width="349" + height="160" + src="https://bottosson.github.io/img/oklab/cielab_blend.png" +/> + +<p> + But all is saved, because one day Björn Ottosson said, "let there be OKLAB". + OKLAB was first described in a blog post he made. It's called OKLAB, because + he thought it was just okay<a id="af-1" href="#footnote-1"><sup>1</sup></a>. + Clearly he was wrong, because it didn't take long for OKLAB to become widely + available in Photoshop, web browsers, and game engines. It's not perfect, and + it's not even always better than CIELAB<a id="af-2" href="#footnote-2"><sup>2</sup></a>, + but it's good enough to beat everything else that's out there. +</p> + +<h2>OKLCH</h2> + +<p> + I don't think you need me to tell you that LAB is hard to visualize. Just guess + what <code>oklab(0.78 0.12 -0.11)</code> is. So, OKLCH came in. It's very + similar to HSL. H and L still stand for hue and lightness, respectively. C + stands for chroma, which is technically different from saturation, but similar + enough that I just think of them as the same thing. +</p> + +<p> + One big difference between OKLCH and HSL is that OKLCH doesn't recognize every + color as being valid. Some colors are simply impossible. Try imagining a green + that's as dark as black, but still clearly green. And there are colors which + exist but cannot be displayed by your monitor, like the brightness of the sun. + OKLCH doesn't attempt to represent them. You can clamp them down to a color + from the sRGB color space if you wish, but there are colors in OKLCH that + cannot be displayed by your monitor, or any monitor for that matter. +</p> + +<p> + I think it's better this way. Some designers may decide to clamp the color into + sRGB anyway, but it's good to at least know that the selected color value + doesn't quite match reality. And on the bright side, it might become slightly + easier to support HDR displays. +</p> + +<p> + If you want to play around with OKLCH, there's a convenient URL for + <a href="https://oklch.com">oklch.com</a>. I recommend giving it a try, + and thinking about it the next time you need a set of colors for your project. +</p> + +</article> +<hr /> +<footer> + <ol> + <li> + <a href="https://qoiformat.org/">QOI</a> uses a similar naming scheme. + <a class="return" href="#af-1">return</a> + </li> + <li> + <a href="https://codepen.io/eeeps/pen/oNmERrJ">https://codepen.io/eeeps/pen/oNmERrJ</a> + <a class="return" href="#af-2">return</a> + </ol> +</footer> +</body diff --git a/src/blog/comments-at-end.kuht b/src/blog/comments-at-end.kuht new file mode 100755 index 0000000..44cdc86 --- /dev/null +++ b/src/blog/comments-at-end.kuht @@ -0,0 +1,662 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Comments at the End</title> + <meta name="description" content="My personal commenting style" /> +</head> + +<body> + +<article> + +<h1>Comments at the End</h1> + +<p> + I'm about to say something which many of you will find comically stupid. I imagine a reaction of both + cringe and shock. Worse, I don't think my rationale is convincing on its own. But I want you to hear + me out, give it a try one day, and maybe I will convince you. If you go through this process and are + not convinced of the benefits, that's fine. As long as you consider it, I will be happy. +</p> + +<p> + This year, I made a new year's resolution: don't write comments until the end, preferably right + before you make the pull request. It's fine to make todo comments, and if you think you might forget + to document something later, it's okay to write it immediately. But at the end, you should go through + your code and write comments. +</p> + +<p> + As is typically recommended, your comments should not explain what your code does. Your code should + explain what your code does. Instead, the comments should explain why you did what you did, and why + you did it in that particular way. Basically, imagine you're showing a co-worker your code, and + explaining what you did. Your comments should reflect everything that you would normally say. Code + that is very simple won't need comments at all. More complicated logic will. +</p> + +<p> + The first benefit to this is that you will be in "writing mode" instead of "coding mode". You'll + discover many points where you need a comment, that you never thought about before. Your comments + will also be very in-depth. I said before that comments should reflect what you would tell someone + when explaining your code. If your comments are this detailed, then you'll never have to explain + anything. This might not work very well if you hate writing. I obviously enjoy it (when I'm not + being graded). I don't know if this benefit will be as big for others. +</p> + +<p> + The bigger benefit, in my opinion, is that it might give you an opportunity to improve your code. I + have another rule, which says, "no apology comments". If you find yourself saying, "Sorry for not + making this better", then you should just fix the problem. Note that you are allowed to say "I didn't + do it this different way because that would be worse". If your solution is the best solution you can + think of, then you <em>should</em> use that solution. But when there's a better alternative, you + should choose the better alternative. As I'm writing comments, I often catch myself in the middle of + writing an apology comment. Then I go fix the problem. That means the commenting phase of my code is + also a minor refactoring phase. +</p> + +<p> + One person has criticized my commenting style, saying that it takes too long to read, and should be + more concise. I think there is a middle ground here, that I haven't achieved yet. In the future, I + might try doing a second pass over my comments to shorten them if I can. I'll just have to hope that + over time, I learn how to make my comments concise from the start. This person has also recommended + that I read Steven Pinker's <em>The Sense of Style</em>. I have a copy of it, but unfortunately I + enjoy writing more than I enjoy reading. +</p> + +<p> + I'm sure some people are still unconvinced, which is reasonable. So I'll provide, below, a sample of + code where I did this by accident, which convinced me to turn this into a habit. If you like what you + see, then I suggest you give my idea a try. This is a C library that allows the + <a href="http://www.fierz.ch/checkerboard.php">CheckerBoard UI</a> to use my + <a href="https://github.com/botahamec/ampere">Ampere Checkers AI</a>. +</p> + +<pre> +// File: ampere_cb.c +// Description: Provides a bridge between Ampere and CheckerBoard +// Author: Mica White (they/them) <botahamec@outlook.com> +// License: CC0 1.0 + +// -------------------------- IMPORTS ------------------------------------ + +#include <math.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <windows.h> + +#include "ampere.h" +#include "cb_interface.h" + + +// --------------------------- MACROS ------------------------------------- + +// The engine should never search to a depth greater than 40 +// This number is somewhat arbitrary. I just doubled the length of the +// shortest possible checkers game +#define MAX_DEPTH 40 +#define MEGABYTE (1024*1024) +// I don't think anybody will ever miss a single megabyte of RAM. +// For computers more powerful than a Raspberry Pi Zero, I recommend +// increasing this to eight megabytes. +#define DEFAULT_HASH_SIZE MEGABYTE + +// These are the lengths of buffers which are given to us by CheckerBoard +#define INFO_STR_LEN 255 +#define COMMAND_LEN 256 +#define CMD_REPLY_LEN 256 +#define NAME_LEN 250 + +// I'm not actually sure when I should start calling this 0.2 +// I'm still planning on making breaking changes, so I won't call it 1.0 +#define ENGINE_NAME "Ampere 0.1" +#define ENGINE_ABOUT \ +"\ +Ampere version 0.1\n\ +A fast and smart Checkers engine\n\ +Implements: transposition table, aspiration windows, iterative deepening\n\ +Defecits: simple evaluation function, no quiescence search\n\ +Copyright Mica White\n\ +" + + +// -------------------------- STATICS ------------------------------------ + +/// The engine itself +// I ended up leaking this. I don't think it'll ever be possible for two +// threads to access the transposition table at the same time, but it's +// probably better safe than sorry. I don't actually know when this would +// be collected anyway. +ampere_engine_t ampere_engine = NULL; +/// A collection of C functions which the engine may call +// This needs to be static because it must outlive `ampere_engine` +struct frontend frontend; +/// The size to set the transposition table to. This can only be modified +/// before the engine is constructed, not after. +int hash_size = DEFAULT_HASH_SIZE; +/// This is a buffer that CheckerBoard provides to display to the user. +// This is static because it needs to be accessed by the `give_info` function. +char* info_str; + + +// --------------------- FRONTEND CALLBACKS ------------------------------- +// These are functions which are passed into the engine on certain events, +// such as logging debug information, displaying evaluation progress, and +// determining the best move. The only useful one for our purpose is +// the `give_info` function, which is called occasionally by the engine +// during its evaluation to give progress reports. + +/// The name says it all +// This is used for both `debug` and `best_move`. We don't do anything +// with debug information because CheckerBoard gives us no way to +// display it. We don't need to do anything with `best_move` because +// our evaluation runs synchronously, and the evaluation function already +// returns the best move. +void do_nothing(void* x) {} + +/// Displays information about the in-progress evaluation to the user +void give_info(ampere_evalinfo_t info) { + char move_str[6] = "None"; // If no move was found, default to "None" + ampere_move_t move = ampere_evalinfo_bestmove(info); + if (move != NULL) { + ampere_move_string(move, move_str); + } + + float eval = ampere_eval_tofloat(ampere_evalinfo_evaluation(info)); + int depth = ampere_evalinfo_depth(info); + unsigned long long int nodes_per_sec = ampere_evalinfo_nodespersec(info); + + sprintf_s( + info_str, INFO_STR_LEN, + "Move: %s; Eval: %f; Depth: %d; NPS: %llu", + move_str, eval, depth, nodes_per_sec + ); +} + + +// ---------------------- HELPER FUNCTIONS -------------------------------- + +/// Converts a CheckerBoard piece to an Ampere piece. +/// This results in undefined behavior for: null pointers, +/// `i` or `j` >= 8, or `index` >= 32, +void square_convert( + Board8x8 board, + int i, + int j, + uint32_t* pieces, + uint32_t* colors, + uint32_t* kings, + int index +) { + int value = board[i][j]; + uint32_t piece_mask = 1 << index; + + if (value == CB_FREE) { + return; + } + else { + *pieces |= piece_mask; + } + + if (value & CB_BLACK) { + *colors |= piece_mask; + } + + if (value & CB_KING) { + *kings |= piece_mask; + } +} + +/// Converts an Ampere piece to a CheckerBoard piece. +/// This results in undefined behavior for a null board or `index` >= 32 +int get_cb_piece(ampere_board_t amp_board, int index) { + int piece = 0; + + if (!ampere_board_has_piece_at(amp_board, index)) { + return CB_FREE; + } + + if (ampere_board_color_at(amp_board, index) == DARK) { + piece |= CB_BLACK; + } + else { + piece |= CB_WHITE; + } + + if (ampere_board_king_at(amp_board, index)) { + piece |= CB_KING; + } + else { + piece |= CB_MAN; + } + + return piece; +} + +/// Converts an ampere square index to a CheckerBoard square coordinate. +/// This is undefined behavior for `amp_square` >= 32 +struct coor get_cb_square(int amp_square) { + struct coor coordinates; + + switch (amp_square) { + case 0: + case 6: + case 12: + case 18: + coordinates.y = 0; + break; + case 1: + case 7: + case 13: + case 19: + coordinates.y = 1; + break; + case 8: + case 14: + case 20: + case 26: + coordinates.y = 2; + break; + case 9: + case 15: + case 21: + case 27: + coordinates.y = 3; + break; + case 16: + case 22: + case 28: + case 2: + coordinates.y = 4; + break; + case 17: + case 23: + case 29: + case 3: + coordinates.y = 5; + break; + case 24: + case 30: + case 4: + case 10: + coordinates.y = 6; + break; + case 25: + case 31: + case 5: + case 11: + coordinates.y = 7; + break; + default: + // this is undefined behavior + break; + } + + switch (amp_square) { + case 18: + case 26: + case 2: + case 10: + coordinates.x = 0; + break; + case 19: + case 27: + case 3: + case 11: + coordinates.x = 1; + break; + case 12: + case 20: + case 28: + case 4: + coordinates.x = 2; + break; + case 13: + case 21: + case 29: + case 5: + coordinates.x = 3; + break; + case 6: + case 14: + case 22: + case 30: + coordinates.x = 4; + break; + case 7: + case 15: + case 23: + case 31: + coordinates.x = 5; + break; + case 0: + case 8: + case 16: + case 24: + coordinates.x = 6; + break; + case 1: + case 9: + case 17: + case 25: + coordinates.x = 7; + break; + default: + // this is undefined behavior + break; + } + + return coordinates; +} + +/// Converts a CheckerBoard board to an Ampere board. +/// This is undefined behavior for a null `board`, or invalid `turn`. +/// It is the caller's responsibility to free the return value using the +/// `ampere_board_destroy` function. Otherwise this results in a memory leak. +ampere_board_t cb_board_to_ampere_board(Board8x8 board, int turn) { + uint32_t pieces = 0l; + uint32_t colors = 0l; + uint32_t kings = 0l; + enum color ampere_turn = DARK; + + square_convert(board, 0, 0, &pieces, &colors, &kings, 18); + square_convert(board, 2, 0, &pieces, &colors, &kings, 12); + square_convert(board, 4, 0, &pieces, &colors, &kings, 6); + square_convert(board, 6, 0, &pieces, &colors, &kings, 0); + square_convert(board, 1, 1, &pieces, &colors, &kings, 19); + square_convert(board, 3, 1, &pieces, &colors, &kings, 13); + square_convert(board, 5, 1, &pieces, &colors, &kings, 7); + square_convert(board, 7, 1, &pieces, &colors, &kings, 1); + square_convert(board, 0, 2, &pieces, &colors, &kings, 26); + square_convert(board, 2, 2, &pieces, &colors, &kings, 20); + square_convert(board, 4, 2, &pieces, &colors, &kings, 14); + square_convert(board, 6, 2, &pieces, &colors, &kings, 8); + square_convert(board, 1, 3, &pieces, &colors, &kings, 27); + square_convert(board, 3, 3, &pieces, &colors, &kings, 21); + square_convert(board, 5, 3, &pieces, &colors, &kings, 15); + square_convert(board, 7, 3, &pieces, &colors, &kings, 9); + square_convert(board, 0, 4, &pieces, &colors, &kings, 2); + square_convert(board, 2, 4, &pieces, &colors, &kings, 28); + square_convert(board, 4, 4, &pieces, &colors, &kings, 22); + square_convert(board, 6, 4, &pieces, &colors, &kings, 16); + square_convert(board, 1, 5, &pieces, &colors, &kings, 3); + square_convert(board, 3, 5, &pieces, &colors, &kings, 29); + square_convert(board, 5, 5, &pieces, &colors, &kings, 23); + square_convert(board, 7, 5, &pieces, &colors, &kings, 17); + square_convert(board, 0, 6, &pieces, &colors, &kings, 10); + square_convert(board, 2, 6, &pieces, &colors, &kings, 4); + square_convert(board, 4, 6, &pieces, &colors, &kings, 30); + square_convert(board, 6, 6, &pieces, &colors, &kings, 24); + square_convert(board, 1, 7, &pieces, &colors, &kings, 11); + square_convert(board, 3, 7, &pieces, &colors, &kings, 5); + square_convert(board, 5, 7, &pieces, &colors, &kings, 31); + square_convert(board, 7, 7, &pieces, &colors, &kings, 25); + + if (turn == CB_WHITE) { + ampere_turn = LIGHT; + } + + return ampere_board_new(pieces, colors, kings, ampere_turn); +} + +/// Converts an Ampere board to a CheckerBoard board. +/// This is undefined behavior if `dest` or `src` are null. +void set_cb_board(Board8x8 dest, ampere_board_t src) { + dest[0][0] = get_cb_piece(src, 18); + dest[2][0] = get_cb_piece(src, 12); + dest[4][0] = get_cb_piece(src, 6); + dest[6][0] = get_cb_piece(src, 0); + dest[1][1] = get_cb_piece(src, 19); + dest[3][1] = get_cb_piece(src, 13); + dest[5][1] = get_cb_piece(src, 7); + dest[7][1] = get_cb_piece(src, 1); + dest[0][2] = get_cb_piece(src, 26); + dest[2][2] = get_cb_piece(src, 20); + dest[4][2] = get_cb_piece(src, 14); + dest[6][2] = get_cb_piece(src, 8); + dest[1][3] = get_cb_piece(src, 27); + dest[3][3] = get_cb_piece(src, 21); + dest[5][3] = get_cb_piece(src, 15); + dest[7][3] = get_cb_piece(src, 9); + dest[0][4] = get_cb_piece(src, 2); + dest[2][4] = get_cb_piece(src, 28); + dest[4][4] = get_cb_piece(src, 22); + dest[6][4] = get_cb_piece(src, 16); + dest[1][5] = get_cb_piece(src, 3); + dest[3][5] = get_cb_piece(src, 29); + dest[5][5] = get_cb_piece(src, 23); + dest[7][5] = get_cb_piece(src, 17); + dest[0][6] = get_cb_piece(src, 10); + dest[2][6] = get_cb_piece(src, 4); + dest[4][6] = get_cb_piece(src, 30); + dest[6][6] = get_cb_piece(src, 24); + dest[1][7] = get_cb_piece(src, 11); + dest[3][7] = get_cb_piece(src, 5); + dest[5][7] = get_cb_piece(src, 31); + dest[7][7] = get_cb_piece(src, 25); +} + + +// --------------------- LIBRARY FUNCTIONS ------------------------------- + +/// Returns the name of the engine +// This shouldn't be necessary since we also implemented the "name" command, +// but this function is very easy to implement, so why not just do it? +int enginename(char reply[NAME_LEN]) { + sprintf_s(reply, NAME_LEN, ENGINE_NAME); + return 1; +} + +/// Responds to an engine command. The following commands are implemented: +/// * name +/// * about +/// * help +/// * set hashsize (only valid before getmove is called) +/// * get hashsize +/// * get protocolversion +/// * get gametype +int enginecommand(char str[COMMAND_LEN], char reply[CMD_REPLY_LEN]) { + // answers to the command sent by CheckerBoard. + char command[COMMAND_LEN] = "", param1[COMMAND_LEN] = "", param2[COMMAND_LEN] = ""; + sscanf_s(str, "%s %s %s", command, COMMAND_LEN, param1, COMMAND_LEN, param2, COMMAND_LEN); + + //by default, return "I don't understand this" + sprintf_s(reply, CMD_REPLY_LEN, "?"); + + if (strncmp(command, "name", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, ENGINE_NAME); + return 1; + } + + if (strncmp(command, "about", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, ENGINE_ABOUT); + return 1; + } + + if (strncmp(command, "help", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, "amperehelp.html"); + return 1; + } + + if (strncmp(command, "set", COMMAND_LEN) == 0) { + if (strncmp(param1, "hashsize", COMMAND_LEN) == 0) { + if (ampere_engine == NULL) { + hash_size = strtol(param2, NULL, 10) * MEGABYTE; + return 1; + } + else { + // we can't change the size of the transposition table after + // the engine is initialized + return 0; + } + } + + if (strncmp(param1, "book", COMMAND_LEN) == 0) { + return 0; + } + } + + if (strncmp(command, "get", COMMAND_LEN) == 0) { + if (strncmp(param1, "hashsize", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, "%d", hash_size / MEGABYTE); + return 1; + } + + if (strncmp(param1, "book", COMMAND_LEN) == 0) { + // theoretically, the correct value is `CB_BOOK_NONE`, but + // CheckerBoard will assume that means: + // "the book hasn't been loaded yet" + // when we actually mean: + // "Ampere doesn't support opening books" + return 0; + } + + if (strncmp(param1, "protocolversion", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, "2"); + return 1; + } + + if (strncmp(param1, "gametype", COMMAND_LEN) == 0) { + sprintf_s(reply, CMD_REPLY_LEN, "%d", GT_ENGLISH); + return 1; + } + } + + return 0; +} + +/// Asks the engine to make a move +int WINAPI getmove( + Board8x8 board, + int color, + double maxtime, + char str[INFO_STR_LEN], + int* playnow, + int info, + int moreinfo, + struct CBmove* move +) { + // this will be modified on every call, just in case CheckerBoard decides + // to reallocate it later + info_str = str; + + if (ampere_engine == NULL) { + // initialize the frontend + frontend.debug = (void (*)(char*)) do_nothing; + frontend.info = give_info; + frontend.report_bestmove = do_nothing; + + // initialize the engine with the frontend and table size + ampere_engine = ampere_new_engine(hash_size, &frontend); + } + + int current_turn = DARK; + if (color == CB_WHITE) { + current_turn = LIGHT; + } + + // set up Ampere's clock + int increment_info = (info >> 2) & 3; + ampere_clock_t clock; + if (increment_info != 0) { + int inc_factor = 1; + if (increment_info == 2) { + inc_factor = 10; + } + else if (increment_info == 3) { + inc_factor = 100; + } + + int time_remaining = ((moreinfo >> 16) & 0xffff) * inc_factor; + int increment = (moreinfo & 0xffff) * inc_factor; + clock = ampere_clock_incremental(time_remaining, time_remaining, increment, increment, 0, 0); + } + else { + int time = floor(maxtime * 1000.0); + clock = ampere_clock_timepermove(time); + } + + ampere_board_t amp_board = cb_board_to_ampere_board(board, color); + ampere_set_position(ampere_engine, amp_board); // the board might change between calls + + struct eval_result result = ampere_evaluate(ampere_engine, (bool*)playnow, 0, MAX_DEPTH, clock); + ampere_move_t best = result.best_move; + ampere_eval_t evaluation = result.evaluation; + + // we calculate the result at the beginning, in case we need to return early + int eval = CB_UNKNOWN; + if (ampere_eval_is_force_win(evaluation)) { + eval = CB_WIN; + } + else if (ampere_eval_is_force_loss(evaluation)) { + eval = CB_LOSS; + } + + // the move will be null if there are no legal moves... + // which means we lost, ergo we could've returned when we got the eval + // i don't really wanna change it now + if (best == NULL) { + return eval; + } + + // I don't think this is actually necessary. CheckerBoard is usually able + // to figure this out automatically for American Checkers. I didn't learn + // that until after I wrote all this code, so I'll leave it in case it + // prevents a bug later. + move->jumps = 0; + move->from = get_cb_square(ampere_move_start(best)); + move->oldpiece = get_cb_piece(amp_board, ampere_move_start(best)); + if (ampere_move_is_jump(best)) { + move->del[move->jumps] = get_cb_square(ampere_move_jump_position(best)); + move->delpiece[move->jumps] = get_cb_piece(amp_board, ampere_move_jump_position(best)); + move->jumps = 1; + } + + // we figure out the new board position, in case we need to double jump + ampere_play_move(ampere_engine, best); + ampere_board_destroy(amp_board); + amp_board = ampere_current_position(ampere_engine); + + // Ampere and CheckerBoard don't align in their definition of a move + // In Ampere, a move is a single slide or jump. But in CheckerBoard, + // double jumps are considered to be one move. This is a very sensible + // decision. But Ampere is very fast partially because it can optimize + // the move struct down to a single byte. So in order for it to work with + // CheckerBoard, we need to go down the line and return any extra jumps + while (*ampere_board_turn(amp_board) == current_turn) { + move->path[move->jumps] = get_cb_square(ampere_move_end(best)); + + // the move should be in the transposition table, so it's ok to set the depth to 1 + result = ampere_evaluate(ampere_engine, (bool*)playnow, 0, 1, 0); + ampere_move_destroy(best); + best = result.best_move; + if (best == NULL) { + return eval; + } + + if (ampere_move_is_jump(best)) { + move->del[move->jumps] = get_cb_square(ampere_move_jump_position(best)); + move->delpiece[move->jumps] = + get_cb_piece(amp_board, ampere_move_jump_position(best)); + move->jumps += 1; + } + + ampere_play_move(ampere_engine, best); + ampere_board_destroy(amp_board); + amp_board = ampere_current_position(ampere_engine); + } + + move->to = get_cb_square(ampere_move_end(best)); + move->newpiece = get_cb_piece(amp_board, ampere_move_end(best)); + + // this is the part that CheckerBoard actually relies upon to figure out what move was played + set_cb_board(board, amp_board); + + ampere_move_destroy(best); + ampere_eval_destroy(evaluation); + ampere_board_destroy(amp_board); + return eval; +} +</pre> + +</article> +</body> diff --git a/src/blog/datetime.kuht b/src/blog/datetime.kuht new file mode 100644 index 0000000..38bceff --- /dev/null +++ b/src/blog/datetime.kuht @@ -0,0 +1,518 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Comparing Date Types Across Languages</title> + <meta name="description" content="Every language has some way of representing time. Some of them are better than others." /> +</head> + +<body> +<article> + +<h1>Comparing Date Types Across Languages</h1> + +<p> +Every language uses a different API to represent types. To put it mildly, some +of them are better than others. This post will compare the best and worst APIs, +for both informative and entertainment purposes. +</p> + +<h2>C</h2> + +<p> +C has multiple ways of representing time, depending on which version you use, +and what operating system you have. I'm not going to bother to look at the +<a href="https://learn.microsoft.com/en-us/windows/win32/api/windows.foundation/ns-windows-foundation-datetime">Win32</a> +or <a href="https://www.man7.org/linux/man-pages/man2/gettimeofday.2.html">POSIX</a> +APIs. Instead I'll focus on plain, simple <code><time.h></code>. +</p> + +<p> +C89 has the `time` function, which returns a +<a href="https://cppreference.com/w/c/chrono/time_t.html"><code>time_t</code></a>. +The specification doesn't say what the type looks like, but it's usually an +integer counting the number of seconds since the UNIX +epoch<a id="af-1" href="#footnote-1"><sup>1</sup></a>. +</p> + +<p> +On its own, this isn't very useful. How would you get information like the +current year, or the current hour? Luckily, C also provides the +<a href="https://cppreference.com/w/c/chrono/gmtime.html"><code>gmtime</code></a> +and <a href="https://cppreference.com/w/c/chrono/localtime.html"><code>localtime</code></a> +functions, which convert the <code>time_t</code> into a +<a href="https://cppreference.com/w/c/chrono/tm.html"><code>tm</code></a>. +</p> + +<pre> +struct tm { + int tm_sec; // seconds after the minute [0, 61?] + int tm_min; // minutes after the hour [0, 59] + int tm_hour; // hours since midnight [0, 23] + int tm_mday; // day of the month [1, 31] + int tm_mon; // months since January [0, 11] + int tm_year; // years since 1900 + int tm_wday; // days since Sunday [0, 6] + int tm_yday; // days since January 1st [0, 365] + // positive is Daylight Savings Time is in effect, + // zero if not, negative if unknown + int tm_isdst; +}; +</pre> + +<p> +There was a mistake in the initial specification where you were allowed to have +62 seconds in a minute. They remembered that leap +seconds<a id="af-2" href="#footnote-2"><sup>2</sup></a> exist, but forgot +about inclusive ranges. This was fixed in C11. +</p> + +<p> +There's also no way to represent either just the date or just the time. If you +want to do either of those things, you'll either have to set some arbitrary day, +or define your own structure. +</p> + +<p> +There's also no time zone information whatsoever here. There's the two +functions to create a <code>tm</code> for the local timezone and UTC, but it's +impossible to tell, just by looking at this structure, which of the two +functions were used to create it. They did, however, include the Daylight +Savings Time information as a nullable boolean. +</p> + +<p> +Overall, I'm not a big fan of this. +</p> + +<h2>JavaScript</h2> + +<p> +JavaScript's date-time class is called <a href="https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Global_Objects/Date"><code>Date</code></a>, +despite also holding information about time. It was copied almost directly from +Java's <a href="https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/util/Date.html"><code>Date</code></a> +class, which was almost entirely obsoleted by JDK 1.1 with the +<a href="https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/util/Calendar.html"><code>Calendar</code></a> +type, for good reason. +</p> + +<p> +Let's start with the constructor. Here it is, according to MDN: +</p> + +<pre> +new Date() +new Date(value) +new Date(dateString) +new Date(dateObject) + +new Date(year, monthIndex) +new Date(year, monthIndex, day) +new Date(year, monthIndex, day, hours) +new Date(year, monthIndex, day, hours, minutes) +new Date(year, monthIndex, day, hours, minutes, seconds) +new Date(year, monthIndex, day, hours, minutes, seconds, milliseconds) + +Date() +</pre> + +<p> +Note that calling `Date()` without the <code>new</code> keyword is equivalent +to <code>new Date().toString()</code>. I wonder how many bugs have been caused +by accidentally passing a string into the constructor instead of a number. +</p> + +<p> +If you pass a year in the range of <code>[0, 99]</code>, the year will be +translated into the 20th century. You might think, "Oh, is that because the +<code>Date</code> class doesn't support dates from that long ago?" No. This +class supports flawless millisecond precision between the years from +271,822 BCE to 175,760 CE, but cannot construct an object for 67 CE without a +workaround. +</p> + +<p> +You might also wonder why MDN uses the term, <code>monthIndex</code> instead +of just, you know, <code>month</code>. Well, every other field looks natural. +For example, January 1st uses <code>1</code> for the day. But in the case of +month, <code>1</code> represents February, and <code>0</code> represents +January. So anyone who naively writes the following will have a bug. +</p> + +<pre> +`${date.getMonth()}/${day.getDay()}/${day.getFullYear()}` +</pre> + +<p> +And now I've just opened another can of worms. What is that +<code>getFullYear</code> method? JavaScript does have a <code>getYear</code> +method, but it's deprecated. It returns the year, minus 1900. The year 2025 +will be returned as 125. The year 1812 is returned as -88. +</p> + +<p> +You may be wondering if that constructor with the year, month, day, hours, +minutes, seconds and milliseconds has the ability to select a timezone. It does +not. In fact, <code>Date</code> contains no timezone information whatsoever in +the object itself. It will always be local time. If you want UTC, you can use +<code>new Date(Date.UTC(year, monthIndex, day, hours, minutes, seconds, milliseconds))</code>. +</p> + +<p> +Because of all of these problems, many JavaScript users decide to ignore the +built-in <code>Date</code> API entirely, and use a library like +<a href="https://momentjs.com/">moment</a>, which will usually add at least 18 +KB to your site's download size, which is bigger than +<a href="https://www.solidjs.com/">some web frameworks</a>. +</p> + +<p> +More recently, we've been granted the Temporal API, but it's not available in +all browsers yet, and the specification is still a draft. I won't get into the +specifics right now in case some of this information becomes out of date. But +there will probably be multiple classes, including <code>Duration</code>, +<code>Instant</code>, <code>PlainDate</code>, and <code>ZonedDateTime</code>. +These types will support nanosecond precision, and the <code>ZonedDateTime</code> +should support proper timezones like <code>"Asia/Shanghai"</code>. +</p> + +<h2>C#</h2> + +<p> +This is the language that inspired me to write this blog post. +</p> + +<p> +In the first versions of .NET, there were two structures related to date and +time: <a href="https://learn.microsoft.com/en-us/dotnet/api/system.datetime?view=net-10.0"><code>DateTime</code></a>, +and <a href="https://learn.microsoft.com/en-us/dotnet/api/system.datetimeoffset?view=net-10.0"><code>DateTimeOffset</code></a>. +These structures, unlike the ones we've seen so far, are pretty well named. +<code>DateTime</code> includes a date and a time. If you need a timezone, +<code>DateTimeOffset</code> includes the date, time, and timezone offset. More +on that later. +</p> + +<p> +There is also a <code>DayOfWeek</code> enum. Perfect! This is probably the best +way of representing days of the week<a id="af-3" href="#footnote-3"><sup>3</sup></a>. +</p> + +<p> +Let's look at some constructors: +</p> + +<pre> +DateTime(int year, int month, int day); +DateTime(int year, int month, int day, int hour, int minute, int second); +DateTime(int year, int month, int day, int hour, int minute, int second, int millisecond); +DateTime(int year, int month, int day, int hour, int minute, int second, int millisecond, DateTimeKind kind); +</pre> + +<p> +Not too bad, but I've omitted some overloads for brevity. Since the language is +statically typed, there's no way to accidentally pass in a string, and there's +no overload for a string. Instead you would use the static method, +<code>DateTime.Parse(string)</code>. +</p> + +<p> +Now you might wonder what that +<a href="https://learn.microsoft.com/en-us/dotnet/api/system.datetimekind?view=net-10.0"><code>DateTimeKind</code></a> +is. Remember how I said that you use <code>DateTimeOffset</code> for timezones? +That's not true. <code>DateTime</code> can also use a timezone, but I don't +think anyone uses it this way. Let's look at the <code>DateTimeKind</code> enum. + +<pre> +enum DateTimeKind { + Unspecified, + Utc, + Local +} +</pre> + +<p> +Hmm. So it can represent timezones, but only the local timezone and UTC. That's +annoying. What if my users are in a different timezone than the server? Tough +luck. +</p> + +<p> +Or no? There's the <code>DateTimeOffset</code> struct. Let's just use that! +Except, as you might have suspected, it is insufficient. The constructors for +<code>DateTimeOffset</code> are similar to the constructors for +<code>DateTime</code>, except for no <code>DateTimeKind</code>, and in their +place is a +<a href="https://learn.microsoft.com/en-us/dotnet/api/system.timespan?view=net-10.0"><code>TimeSpan</code></a> +struct. +</p> + +<pre> +DateTimeOffset(DateTime dateTime, TimeSpan offset); +DateTimeOffset(int year, int month, int day, int hours, int minutes, int seconds, TimeSpan offset); +</pre> + +<p> +That <code>TimeSpan</code> doesn't seem like a very good timezone, and indeed +it is not. You can't specify a timezone like "America/New_York". Instead, you +specify `UTC-6`. When New York goes into daylight savings time, then you also +need to update the offset<a id="af-4" href="#footnote-4"><sup>4</sup></a>. +</p> + +<p> +Now we're getting back into the problems with JavaScript's <code>Date</code> +class. There's no timezone information whatsoever, and the timezone information +that we can provide is worse than useless. I'm told that most people who have +to work with time in C# use an external library, but this wasn't the case at the +company I worked at. At least in this case, your users aren't forced to +download the library<a id="af-5" href="#footnote-5"><sup>5</sup></a>. +</p> + +<p> +You may have noticed that, unlike with <code>DateTime</code>, there's no way to +create a <code>DateTimeOffset</code> without specifying the time. There's also +no way to create a <code>DateTime</code> without specifying a date. You could +argue that this is a good thing, since a <em>date time</em> should include both +a <em>date</em> and a <em>time</em>. But for a long time, there was no +alternative. +</p> + +<p> +You might look through the documentation and get excited, because of the +<code>Date</code> property, which presumably returns a new structure I hadn't +mentioned yet which only contains the date information. Unfortunately, this +property is completely useless. It returns the same <code>DateTime</code>, but +with the time set to midnight. There's also a <code>TimeOfDay</code> property +which returns a <code>TimeSpan</code> representing the time that has elapsed +since midnight. +</p> + +<p> +Fortunately, in .NET 6, we got the +<a href="https://learn.microsoft.com/en-us/dotnet/api/system.dateonly?view=net-10.0"><code>DateOnly</code></a> +and +<a href="https://learn.microsoft.com/en-us/dotnet/api/system.timeonly?view=net-10.0"<code>TimeOnly</code></a> +structs. These do exactly what you think they would do. +</p> + +<pre> +DateOnly(int year, int month, int day); +TimeOnly(int hour); +TimeOnly(int hour, int minute); +TimeOnly(int hour, int minute, int second); +TimeOnly(int hour, int minute, int second, int millisecond); +TimeOnly(int hour, int minute, int second, int millisecond, int microsecond); +</pre> + +<p> +There is a caveat here, though. The <code>DateTime.Date</code> property still +doesn't return a <code>DateOnly</code>. A part of me hoped that after these +types were introduced, a breaking change to the language could be made to +replace the completely useless property. Alas, we are stuck with that. +</p> + +<h2>Rust</h2> + +<p> +Of course, I have to talk about Rust. What does Rust do? Let's look at the +<code>time</code> module. It includes three types worth caring about: +<a href="https://doc.rust-lang.org/stable/std/time/struct.Duration.html"><code>Duration</code></a>, +<a href="https://doc.rust-lang.org/stable/std/time/struct.Instant.html"><code>Instant</code></a>, and +<a href="https://doc.rust-lang.org/stable/std/time/struct.SystemTime.html"><code>SystemTIme</code></a>. +The behavior of <code>Duration</code> should be obvious, but you may wonder +what <code>Instant</code> and <code>SystemTime</code> are. Let's start with +<code>SystemTime</code>. +</p> + +<pre> +pub struct SystemTime(/* private fields */); + +impl SystemTime { + const UNIX_EPOCH: SystemTime; + + fn now() -> SystemTime; + fn duration_since(&self, earlier: Self) -> Result<Duration>; + fn elapsed() -> Result<Duration>; + fn checked_add(&self, duration: Duration) -> Option<Self>; + fn checked_sub(&self, duration: Duration) -> Option<Self>; +} +</pre> + +<p> +And that's it! What? You were expecting more? This is every method implemented +on <code>SystemTime</code> outside of traits. No formatting, no figuring out +the current year, just that. +</p> + +<p> +Ok, surely <code>Instant</code> must be more useful, right? Nope. It's actually +the same as <code>SystemTime</code>, except it is monotonically +increasing<a id="af-6" href="#footnote-6"><sup>6</sup></a>. What is this? +</p> + +<p> +The Rust standard library is small, on purpose. They don't include features +unless the developers are confident in both the API and its utility. The other +languages in this post should make it obvious that this is a difficult feature +to make a good API for. So it's better to not include dates and times in the +standard library, and just let external libraries handle that. +</p> + +<p> +On the other hand, many low-level system APIs do require some time information. +For example, the <code>Metadata</code> struct contains the time when a file was +last modified. So, there needs to be an <code>Instant</code> struct, but it is +very small, and mostly just a wrapper around the values used by the system +calls. +</p> + +<p> +That being said, I do want to talk about a Rust library that I personally like. +My favorite is <a href="https://docs.rs/chrono/latest/chrono/index.html"><code>chrono</code></a>. +I mostly just want to talk about the +<a href="https://docs.rs/chrono/latest/chrono/struct.DateTime.html"><code>DateTime</code></a> +type. Needless to say, it has much more functionality than the +<code>SystemTime</code> type, so I won't go over all of it. But I do want to +show the declaration. +</p> + +<pre> +struct DateTime<Tz: TimeZone> { + datetime: NaiveDateTime, + offset: Tz::Offset. +} +</pre> + +<p> +That's different. You might correctly guess that +<a href="https://docs.rs/chrono/latest/chrono/trait.Offset.html"><code>NaiveDateTime</code></a> +is just a date and a time with no timezone information. But what's that +<a href="https://docs.rs/chrono/latest/chrono/trait.TimeZone.html"><code>TimeZone</code></a> trait? +</p> + +<pre> +trait TimeZone: Sized + Clone { + type Offset: Offset; + + fn from_offset(offset: &Self::Offset) -> Self; + fn offset_from_local_date(&self, local: &NaiveDate) -> MappedLocalTime<Self::Offset>; + fn offset_from_local_datetime(&self, local: &NaiveDateTime) -> MappedLocalTime<Self::Offset>; + fn offset_from_utc_date(&self, utc: &NaiveDate) -> Self::Offset; + fn offset_from_utc_datetime(&self, utc: &NaiveDateTime) -> Self::Offset; +} + +trait Offset: Sized + Clone + Debug { + fn fix(&self) -> FixedOffset; +} + +enum MappedLocalTime<T> { + Single(T), + Ambiguous(T, T), + None, +} +</pre> + +<p> +This is far more complex than the timezone representation in C#. We do see the +<a href="https://docs.rs/chrono/latest/chrono/trait.Offset.html"><code>Offset</code></a> +trait in there, but there's more to it than that. The <code>TimeZone</code> +trait includes several methods for getting the offset from UTC for a given date +and time. So, we can have different offsets at different dates. Finally, we can +transparently handle daylight savings time for timezones other than the local +timezone! And since the timezone is a generic type, we can easily infer from +the types what the timezone is going to be, rather than having to look at how +the object was constructed. +</p> + +<p> +The <code>chrono</code> crate by default includes three timezones: +<a href="https://docs.rs/chrono/latest/chrono/struct.FixedOffset.html"><code>FixedOffset</code></a>, +<a href="https://docs.rs/chrono/latest/chrono/struct.Local.html"><code>Local</code></a>, +and <a href="https://docs.rs/chrono/latest/chrono/struct.Utc.html"><code>Utc</code></a>. +This is already as good as what we were provided in C#. But remember that +<code>TimeZone</code> is a trait that we can implement ourselves. I recommend +importing the <a href="https://docs.rs/chrono-tz/0.10.4/chrono_tz/"><code>chrono-tz</code></a> +crate, which includes every time zone under the sun. There's also a generic +<a href="https://docs.rs/chrono-tz/0.10.4/chrono_tz/enum.Tz.html"><code>Tz</code></a> +enum, which can represent any timezone if you need it. The implementors of the +trait need not be empty structs. +</p> + +<p> +This is, by far, the best implementation of a time API I've seen anywhere. I'm +sure there are libraries for other languages which do the same thing, and I +recommend trying them out. +</p> + +<p> +The downside to <code>chrono</code> is that, at time of writing, it's +unmaintained. Hopefully a new maintainer will take it over some day soon. For +now, <a href="https://docs.rs/jiff/latest/jiff"><code>jiff</code></a> is the +most popular maintained time crate for Rust. It takes heavy inspiration from +JavaScript's new Temporal API that we talked about earlier. It's no chrono, +but it gets the job done, and they're approaching a 1.0 release. +</p> + +<h2>Conclusion</h2> + +<p> +My conclusion is my own opinion. You may have one that differs from mine. But +here's what I like to see: +</p> + +<ul> + <li>Handline of timezones</li> + <li>The timezones cannot be plain offsets from UTC</li> + <li>Even better: have an implementable `TimeZone` interface that records the timezone information in the type</li> + <li>Enums for days of the week and months are great</li> + <li>When passing in numbers as months, use 1 for January</li> + <li>Higher resolution is better</li> +</ul> + +<p> +Hopefully this will inspire you to either go out and see what other time +libraries are out there, or make one yourself. +</p> + +</article> + +<hr /> + +<footer> + <ol> + <li id="footnote-1"> + The <a href="https://en.wikipedia.org/wiki/Unix_time">UNIX epoch</a> is + midnight, January 1st, 1970. <a class="return" href="#af-1">return</a> + </li> + <li id="footnote-2"> + A <a href="https://en.wikipedia.org/wiki/Leap_second">leap second</a> is when + a minute contains 61 seconds. This is done from time to time to account for + the slowing down of the rotation of the Earth. Unlike leap years, which + happen at predictable times, leap seconds are decided by commitee. This + makes them difficult to work with on computers, and most timestamps ignore + them, even though the time format internally allows them to be represented. + <a class="return" href="#af-2">return</a> + </li> + <li id="footnote-3"> + It's better if you use a language with good enums. C# enums are aliases for + integers, which makes them somewhat less useful. + <a class="return" href="#af-3">return</a> + </li> + <li id="footnote-4"> + I have discovered a bug caused by this before. There were other confounding + factors in that case, but the bug occurred when daylight savings time + changed. <a class="return" href="#af-4">return</a> + </li> + <li id="footnote-5"> + Assuming you're not converting your C# to WebAssembly and sending it to the + browser, which is probably more common with C# than most other languages, + but most people just use JavaScript. + <a class="return" href="#af-5">return</a> + </li> + <li id="footnote-6"> + Monotonic time just means that the time never ever decreases. You might + think that this should always be the case, but computers are complicated. + <a class="return" href="#af-6">return</a> + </li> + </ol> +</footer> + +</body> diff --git a/src/blog/how-happylock-works.kuht b/src/blog/how-happylock-works.kuht new file mode 100755 index 0000000..c1cf631 --- /dev/null +++ b/src/blog/how-happylock-works.kuht @@ -0,0 +1,767 @@ +<import "base.kuht" as "base" /> + +<head> + <title>How HappyLock Works</title> + <meta name="description" content="Recently, I released version 0.3 of a Rust library which prevents deadlocks at compile-time. I want to explain what I changed, and why it works" /> +</head> + +<body> + +<article> + +<h1>How HappyLock Works</h1> + +<p> + Recently, I released version 0.3 of my <a href="https://www.lib.rs/happylock">HappyLock</a> crate + on crates.io. In this blog post, I wanted to explain what I changed, and why it works. +</p> + +<h2>Background</h2> + +<p> + There are four conditions necessary for a deadlock to occur. You only need to prevent one of them + in order to prevent all deadlocks: +</p> + +<ol> + <li>Mutual exclusion</li> + <li>Non-preemptive allocation</li> + <li>Circular wait</li> + <li>Partial allocation</li> +</ol> + +<p>Let's go through each one, and see what we can do.</p> + +<h3>Mutual exclusion</h3> + +<p> + This doesn't make all that much sense to prevent. We could provide some sort of + <code>Readonly<T></code> type, but Rust already has <code>&T</code> and + <code>Arc<T></code>. So this wouldn't be very useful. +</p> + +<h3>Non-preemptive allocation</h3> + +<p> + Preventing this would mean that the language can decide to take away your access to a piece of + data at any time. Not only would this be incredibly difficult to pull off, it would also just + annoying for the programmer to deal with. Something like this would need to happen: +</p> + +<pre> +let mutex = Mutex::new(5); +let mut_ref = &mutex; +let th = thread::spawn(|| { + let number = mut_ref.lock(); + *number = 6; +}); + +let number = mut_ref.lock(); +th.join(); // preempts the lock on mut_ref +println!("{}", number); // oops, we can't use number anymore +</pre> + +<p> + We could have some sort of <code>LockCell</code> type, which releases the lock immediately after + performing some action, but there are often times where you need to make sure nothing else can + mutate the value for a section of code. So this won't always work. +</p> + +<h3>Circular wait</h3> + +<p> + From what I can tell, most attempts to prevent deadlock go through this route. The idea is what + every lock must be acquired <i>in the same order</i>. So if you lock <code>l1</code> and then + <code>l2</code> on one thread, then you can't do it in the opposite order on the other thread. + This is an interesting approach, but there's no way to enforce it in the language. Some people + have resorted to using macros for this, such as in the recently published + <a href="https://www.lib.rs/deadlocker">deadlocker crate</a> (which is a very confusing name). + But an attempt to make this happen at compile-time won't be dynamic enough for the general case. +</p> + +<p>Keep an eye on this one, because we will make use of it later.</p> + +<h3>Partial allocation</h3> + +<p> + This is the hero we've been waiting for. To prevent this, we need to enforce total allocation. So if + you want to lock a new mutex, you must first release the locks you've already acquired, and then + acquire a new set of locks. Preventing this is possible in Rust, and less possible in languages that + don't have a borrow checker. Let me explain how we'll use the borrow checker. +</p> + +<h2>Exploration</h2> + +<p> + Say we were making mutexes for an operating system. How would we enforce total allocation? We could + simply give an error if someone acquires a new set of locks without releasing the already acquired + locks. It'd also mean we'd have to check for an error every time we acquire a new set of locks. +</p> + +<pre> +mutex_t m1 = new_mutex(); +mutex_t m2 = new_mutex(); + +mutex_t mutices[] = { m1, m2 }; +if (lock_mutexes(mutices, 2)) { + // oh noes! +} + +// now we can use the data +</pre> + +<p>But we're not C. We have technology! And our technology is the borrow checker.</p> + +<h3>Borrow Checker Rules</h3> + +<p>For those of you who don't know, here's a quick rundown of the borrow checker rules:</p> + +<ol> + <li>A value may not be moved while references to it exist.</li> + <li>Only one mutable reference to a value may exist at a time.</li> + <li>A mutable reference and immutable references, must not exist at the same time.</li> +</ol> + +<p>As a quick example of these rules:</p> + +<pre> +let s = String::new("Hello, world!"); +let r1 = &s; +let r2 = &s; // this is allowed because these references aren't mutable +let mr = &mut s; // illegal: rule #3 +drop(s); // also illegal: rule #1 +println!("{r1} {r2}"); +</pre> + +<h2>A Naive implementation</h2> + +<p> + So here's the plan: We'll make a new type called <code>ThreadKey</code>. This type cannot be + cloned, copied, sent to another thread, or mutably referenced from another thread. The Rust type + system is able to enforce all of those things. +</p> + +<pre> +// no traits are derived, especially not Clone or Copy +pub struct ThreadKey { + phantom: PhantomData<*const ()>, // !Send and !Sync +} +</pre> + +<p> + Then, we can require that when locking a mutex (or a read-write lock), a <code>ThreadKey</code>, + or a <code>&mut ThreadKey</code>, must be given. +</p> + +<pre> +pub fn lock<'s, 'k: 's, Key: Keyable>(&'s self, key: Key) -> MutexGuard<'_, 'k, T, Key, R> +</pre> + +<p> + Wow! That's a lot of lifetimes and generics. We'll worry about those later. For now, let's show + how this works. +</p> + +<pre> +use happylock::{ThreadKey, Mutex}; + +fn main() { + // each thread can only have one thread key + // the call to unwrap panics if this thread already got its key + // ThreadKey is not Send, Sync, Copy, or Clone + let key = ThreadKey::get().unwrap(); + + let mutex = Mutex::new(10); + + // locking a mutex requires either the ThreadKey or a &mut ThreadKey + let mut guard = mutex.lock(key); + + // we no longer have the thread key, so we can't lock anything else anymore + + println!("{}", *guard); +} +</pre> + +<p> + I recently learned that I wasn't the only person who thought of this. Over a year ago, Adrian + Taylor posted on Medium, + <a href="https://medium.com/@adetaylor/can-the-rust-type-system-prevent-deadlocks-9ae6e4123037"> + Can the Rust type system prevent deadlocks?</a>. + I was completely unaware of this when I published the crate, despite my search for other + implementations of the idea. He also worked on getting inner mutexes to work, which is currently + impossible in HappyLock. He didn't attempt to do something more obvious though, which is locking + more than one mutex at a time. He also never considered the possibility of using a + <code>& mut MutexPermissionToken</code> to make the code more ergonomic. +</p> + +<p>Speaking of which, how do we lock more than one mutex at a time?</p> + +<h3>Lock Collections</h3> + +<p> + It's actually quite simply, we just lock a collection of things. We can have a type, called + <code>LockCollection</code>, which takes a collection of locks, and locks them all at the same time. + Then, it'll return a <code>LockGuard</code> for the collection. +</p> + +<p> + There are different collections we could try to lock, including <code>Vec<T></code>, + <code>(A, B)</code>, and <code>(Vec<A>, Vec<B>)</code>, so this'll have to be generic. To + enable this, we'll create a trait called <code>Lockable</code>. +</p> + +<p>This isn't the current implementation in HappyLock, but here's a proof-of-concept:</p> + +<pre> +unsafe trait Lockable { + type Guard<'a>; + + unsafe fn lock<'a>(&self) -> Self::Guard<'a>; + + unsafe fn try_lock<'a>(&self) -> Option<Self::Guard<'a>>; +} +</pre> + +<p>Then, we just need to implement on every collection that could hold a lock.</p> + +<pre> +use happylock::{ThreadKey, Mutex, LockCollection}; + +fn main() { + let key = ThreadKey::get().unwrap(); + let mutex1 = Mutex::new(5); + let mutex2 = Mutex::new(String::new()); + let collection = LockCollection::new((mutex1, mutex2)); + let guard = collection.lock(key); + *guard.1 = format!("{}{}", *guard.1, guard.0); + *guard.0 += 1; +} +</pre> + +<p>That seems to work pretty well.</p> + +<h3>Duplicate Locks</h3> + +<p>Let's try something:</p> + +<pre> +use happylock::{ThreadKey, Mutex, LockCollection}; + +fn main() { + let key = ThreadKey::get().unwrap(); + let mutex1 = Mutex::new(5); + // oh no. this will deadlock us + let collection = LockCollection::new((&mutex1, &mutex1)); + let guard = collection.lock(key); +} +</pre> + +<p> + The problem is that we passed the same lock in twice. That'll be a problem if we require it to be + locked twice. The good news is, this code doesn't compile, because <code>LockCollection::new</code> + actually requires a different trait that I haven't mentioned. +</p> + +<pre> +// not implemented for &impl Lockable +// ergo: the values within are guaranteed to be unique +unsafe trait OwnedLockable: Lockable {} +</pre> + +<p> + Let me elaborate on this. If the value is owned, then there's no way to pass it twice into the same + collection. The same is true for mutable references. So we can safely lock an owned lock without + checking for duplicates. +</p> + +<p> + But sometimes, it's useful to lock a referenced mutex. For these, we'll need a runtime check for + duplicates. So we'll need to add another method to our <code>Lockable</code> trait. +</p> + +<pre> +unsafe trait Lockable { + // ... snip ... + + fn get_ptrs(&self) -> Vec<usize>; +} +</pre> + +<p> + Then we can check for duplicate pointers within a collection. My first implementation of that check + was a brute-force O(n²) check. When I posted about this on Reddit, people were quick to point out + that this could be made O(nlogn) by sorting the pointers, then checking if the same pointer appeared + twice in a row. +</p> + +<pre> +fn contains_duplicates<L: Lockable>(data: L) -> bool { + let mut pointers = data.get_ptrs(); + pointers.sort_unstable(); + pointers.windows(2).any(|w| w[0] == w[1]) +} +</pre> + +<p>Of course, you could make this an O(n) check by using a <code>HashSet</code> too.</p> + +<p>Now we need a new constructor for <code>LockCollection</code>.</p> + +<pre> +impl<L: Lockable> LockCollection<L> { + pub fn try_new(data: L) -> Option<Self> { + let ptrs = data.get_ptrs(); + if contains_duplicates(&ptrs) { + return None; + } + + Some(Self { data }) + } +} +</pre> + +<h2>The Actual Implementation</h2> + +<p> + Although we have now successfully prevented deadlocks, + <a href="https://en.wikipedia.org/wiki/Deadlock#Livelock">livelocking</a> can still be an issue. +</p> + +<p> + The first implementation of <code>LockCollection</code> would try to lock everything in order. If + one of the locks blocked, then it would release everything it had and try again. It would keep + retrying until it successfully locked everything in a row. This resulted in a lot of wasted work. It + also made the <code>LockCollection</code> start to resemble a spinlock. This also made the + collection prone to livelocking. +</p> + +<p> + Imagine this scenario with two threads trying to lock a set of mutexes, but in a different order: +</p> + +<ol> + <li>Thread 1 locks mutex 1</li> + <li>Thread 2 locks mutex 2</li> + <li>Thread 1 tries to lock mutex 2 and fails</li> + <li>Thread 2 tries to lock mutex 1 and fails</li> + <li>Thread 1 releases mutex 1</li> + <li>Thread 2 releases mutex 2</li> + <li>Repeat</li> +</ol> + +<p> + In practice, this loop would probably end eventually, but it'd be nice if we could prevent it + altogether. You could avoid this by locking everything in the same order. But if we could do that, + then we would've prevented cyclic wait from before, which would have avoided this whole problem. +</p> + +<h3>Cyclic wait</h3> + +<p> + Remember when I said we would make use of cyclic wait later? This is where we'll do this. If you + recall from earlier, we already sorted the mutex pointers by their memory address. What if we locked + them in that order? This was suggested by many people on Reddit, but it took a while to work out + exactly how to do this. I can't sort a tuple. So we'll need to lock them in a different order than + the one they're returned in. This will require a large change to the existing <code>Lockable</code> + trait. Let's see what we can do. +</p> + +<pre> +unsafe trait Lock { + unsafe fn lock(&self); + unsafe fn try_lock(&self) -> bool; + unsafe fn unlock(&self); +} + +unsafe trait Lockable { + type Guard<'g>; + fn get_locks<'a>(&'a self, &mut Vec<&'a dyn Lock>); + unsafe fn guard<'g>(&'g self) -> Self::Guard<'g>; +} +</pre> + +<p> + <code>Lock</code> is implemented on <code>Mutex</code> and <code>RwLock</code>. <code>Lockable</code> + is implemented on the same types as before. These traits are even more unsafe than they were before, + now that they can obtain a guard without locking it first. So, we'll just need to be very careful. +</p> + +<p>Now let's try making our new <code>LockCollection</code></p> + +<pre> +struct LockCollection<L> { + data: L, + locks: Vec<&'self.data dyn Lock> + // wait, this is a self-referential struct +} +</pre> + +<p> + Ooh, boy. This is gonna be a problem. We could box the data, but sometimes we don't want to pay the + price of memory allocation. We could try using a reference, but then we have an extra lifetime to + deal with. It's also sometimes useful to have an owned lock collection. How are we going to do this + without creating four lock collection types that do mostly the same thing? +</p> + +<h3>We'll just create four Lock Collection Types that do mostly the same thing</h3> + +<p> + I'll cover each of the four types, but if you're looking for a sensible default, there's a type alias + so that <code>LockCollection</code> points to <code>BoxedLockCollection</code>. It's slower and + doesn't allow you to mutate the underlying collection (yet), but it works for most use cases. +</p> + +<p> + I'm pretty sure when this gets posted to Reddit, someone will tell me I could've avoided this + somehow. +</p> + +<h4><code>RefLockCollection</code></h4> + +<pre> +pub struct RefLockCollection<'a, L> { + data: &'a L, + locks: Vec<&'a dyn RawLock>, +} +</pre> + +<p> + It needs to sort the locks by address. Yes, that applies even if the collection is owned. Otherwise + the locks could be acquired in a different order. We're no longer releasing the locks when we can't + acquire them all at once, so they need to be acquired in exactly that order. But it avoids one memory + allocation. +</p> + +<h4><code>BoxedLockCollection</code></h4> + +<pre> +pub struct BoxedLockCollection<L> { + data: *const UnsafeCell<L>, + locks: Vec<&'static dyn RawLock>, +} +</pre> + +<p> + This is similar to <code>RefLockCollection</code>, but it allocates more memory. You may notice that + there's no <code>Box</code> in this structure, despite the name. That's because <code>Box</code> + needs to be a unique pointer. But because of the references in <code>locks</code>, it's not unique. + It's <code>const</code> because of the existing references to the collection, meaning mutating the + underlying collection would cause undefined behavior, especially if it means the locks move (such + as when a <code>Vec</code> reallocates). It's a raw pointer, because deallocating the data requires + a mutable pointer, and creating a mutable pointer from an immutable reference is undefined behavior. + The <code>UnsafeCell</code> is used for a similar reason. +</p> + +<p> + This is actually the reason why this release is titled 0.3 and not 0.2. I released 0.2 and found + these soundness bugs, so I had to make a fix. This was a breaking change, because you can't mutate + the underlying collection anymore. +</p> + +<h4><code>OwnedLockCollection</code></h4> + +<pre> +pub struct OwnedLockCollection<L> { + data: L, +} +</pre> + +<p> + Since both <code>RefLockCollection</code> and <code>BoxedLockCollection</code> require sorting the + locks, it'd be nice if we could sometimes avoid that for owned lock collections. If a set of locks is + in an <code>OwnedLockCollection</code>, then it really cannot be used anywhere else. That means as + long as references to a <code>OwnedLockCollection</code> always lock in the same order, then it'll + never deadlock. That's the only order these particular locks can be locked in. So, we don't bother + sorting and just lock them in the order they appear in for the collection. +</p> + +<h4><code>RetryingLockCollection</code></h4> + +<p> + This is the collection that was used in Happylock 0.1. This does the releasing of all of the locks + thing. The upside of this collection is that it can contain references, and still not sort the + pointers. Duplicates are checked for by using a <code>HashSet</code>. Because of the lack of sorting, + it can be fast in the following case, from the docs: +</p> + +<blockquote> + [...] when the first lock in the collection is always the first in any + collection, and the other locks in the collection are always locked after + that first lock is acquired. This means that as soon as it is locked, there + will be no need to unlock it later on subsequent lock attempts, because + they will always succeed. +</blockquote> + +<h3>RwLocks</h3> + +<p> + It doesn't take a genius to realize that my lock collection API is not friendly to read-write locks. + In HappyLock 0.1, I had <code>ReadLock</code> and <code>WriteLock</code> wrappers for + <code>RwLock</code> to allow reads in collections. +</p> + +<pre> +struct ReadLock<'a, T(&'a RwLock<T>); +struct WriteLock<'a, T(&'a RwLock<T>); +</pre> + +<p> + This worked fine at the time, but I quickly realized that this would not work for an + <code>OwnedLockCollection</code>. To solve this problem, I made more changes to the + <code>Lockable</code> API. +</p> + +<pre> +unsafe trait Lock { + // ... + unsafe fn read(&self); + unsafe fn try_read(&self) -> bool; + unsafe fn unlock_read(&self); +} + +unsafe trait Lockable { + // ... + type ReadGuard<'g>; + unsafe fn read_guard<'g>(&'g self) -> Self::ReadGuard<'g> +} + +// A NEW TRAIT +unsafe trait Sharable: Lockable {} +</pre> + +<p> + The <code>Sharable</code> trait is implemented on collections that only contain readable locks. + All of the lock collection types then have functions for when their locks are all sharable. That + allows us to grant read-access in <code>OwnedLockCollection</code>s. +</p> + +<p>After I was done with that, I updated the documentation and published the new version.</p> + +<h2>Performance</h2> + +<p> + <code>ThreadKey</code> is a zero-cost abstraction. It takes zero space at runtime. The only code is + checking to see if the key has already been acquired for a thread. That is a static, + lazily-initialized, thread local, boolean check and update, so it's fairly fast. +</p> + +<p> + I used <code>lock_api</code> as backends for the <code>Mutex</code> and <code>RwLock</code> types. + By default, this library acts as a thin wrapper around <code>parking_lot</code>, so it will be + roughly the same speed as <code>parking_lot</code>. There's an optional feature that can be enabled + to use <code>spin</code> as well, and you can import your own raw locks if you wish. The standard + library locks cannot be used, because it's impossible to implement the <code>lock_api</code> traits + on them. If someone adds a <code>Mutex::force_unlock</code> function to the standard library, then I + can add that support, and I'll even make it the default. For now, we have to deal with the larger + binary size that's given by <code>parking_lot</code>. +</p> + +<p> + The only place where performance could be an issue is in the lock collection types. The performance + qualities of each type have already been discussed. But to summarize, only the + <code>OwnedLockCollection</code> is zero-cost. Both <code>RefLockCollection</code> and + <code>BoxedLockCollection</code> allocate memory and sort the lock pointers, and sometimes they'll + even need to check for duplicates within the pointers. All of this work is done when the collection + is created though, so at least you don't need to worry about it after that. On the other hand, the + <code>RetryingLockCollection</code>, when locking, will waste time unlocking and relocking unless the + following condition is met: +</p> + +<blockquote> + [...] when the first lock in the collection is always the first in any collection, and the other + locks in the collection are always locked after that first lock is acquired. +</blockquote> + +<p><code>RetryingLockCollection</code> also checks for duplicates using a <code>HashSet</code>.</p> + +<p>In summary, it's pretty fast, but choose the correct lock collection constructor carefully.</p> + +<h2>Future Work</h2> + +<p> + This is summarized on the GitHub repo, but I'll try to go into more depth on my plans for the future + of this library. +</p> + +<h3>Poisoning</h3> + +<p> + Personally, I'm not a big fan of mutex poisoning, but I can see why someone would want to use it. I + plan on adding it to Happylock soon, by having a <code>Poisonable</code> wrapper around + <code>Lock</code> types. And if I implement <code>Lock</code> on the lock collection types, then + it'll even be able to wrap those, which might be useful. +</p> + +<h3>OS Locks</h3> + +<p> + I originally thought of this as being a must-have for the next version of HappyLock, but after trying + it for a bit, I decided to hold off on it. As I mentioned before, there is a binary size cost to + using <code>parking_lot</code>. We could save memory by using the locks that are built into the + operating system, which is what the standard library does. It would be very convenient if I could + just use the standard library as a backend for HappyLock. +</p> + +<p> + Unfortunately, there's a reason why the standard library's locks do not implement + <code>lock_api</code>'s traits. There's just no way to unlock one of those mutexes without holding + onto the guard. If we were to put the guard inside of the mutex, and then drop it when unlocking, + that would make it impossible to share the lock across threads, since the guard isn't + <code>Send</code>. +</p> + +<p> + There are raw mutexes used in the underlying implementation of the standard library, but those aren't + public. I've tried simply copying the part of the standard library that implements those traits, but + it relies on too much of the rest of the standard library. I don't want importing HappyLock to + require recompiling the entire standard library. +</p> + +<p> + The next feasible solution, in my opinion, would be to make a new crate that uses the operating + system mutexes, and implements <code>lock_api</code> traits on them. I've already started doing this, + and it seems very possible. But don't expect it to properly support every platform that the standard + library supports. +</p> + +<h3>Compile-Time Duplicate Checks</h3> + +<p> + The only way to make duplicate checks free is to run them at compile-time. I doubt I'd be able to + implement this using procedural macros (feel free to prove me wrong). But, as + <a href="https://github.com/pacak">pacak</a> keeps reminding me, it might be possible using const + functions with a <a href="https://en.wikipedia.org/wiki/Bloom_filter">Bloom filter</a>. I don't know + if const Rust is robust enough for this to be implemented yet, but I do want to give it a try. It + wouldn't work for lock collections which require sorting, but it might work for + <code>RetryingLockCollection</code>. +</p> + +<h3>Expanding Cyclic Wait</h3> + +<p> + One other thing I'd like to explore is if there's anything more that can be done with cyclic wait. + I could imagine <code>LockCollection<Vec<L>> where L: OwnedLockable</code> having some + <code>lock_next</code> method, or maybe an iterator type that would allow you to only lock the first + few items as they are needed, rather than locking the entire collection at once. I'm not sure how + useful this would be though. I'm sure there are other ideas that I'm not thinking of too. +</p> + +<h3>A <code>LockCell</code> type</h3> + +<p> + If you've seen the <code>Cell</code> type in Rust, then you know that it's a good zero-cost + abstraction that provides safe interior mutability. The downsides are that it cannot be shared + between threads and its limited API. But I think it would be neat to add some of its methods as + convenience methods for <code>Mutex</code> and <code>RwLock</code>, i.e: + <code>Mutex::lock_swap</code>. This would not remove the need for a <code>ThreadKey</code>, since + then you'd be able to call <code>lock_swap</code> on a thread that's already locked the mutex. +</p> + +<p> + However, if we had methods like <code>Mutex::try_swap</code>, then there'd be no need. The mutex + wouldn't be able to block the current thread, and because the lock is released so quickly, it + wouldn't block any other threads either. No blocking, means no deadlock. I think the same would be + true for <code>try_clone</code> and <code>try_take</code>, but they're scarier since the + <code>Clone</code> and <code>Default</code> traits could try to lock the same mutex, resulting in a + deadlock. I don't know if it would be possible to access the <code>ThreadKey</code> safely in these + methods though. Either way, all of these methods would be fallible, so it's not going to be a + perfect solution. +</p> + +<p> + But we could introduce a <code>LockCell</code> type, which would be the same as <code>Cell</code>, + but it safely implements <code>Sync</code> by using a mutex or rwlock internally. This would not + require <code>ThreadKey</code> at all, as long as all of the implemented methods don't lock the + <code>LockCell</code> a second time, which is probably impossible. +</p> + +<p> + It is a little surprising that a type like this doesn't already exist in the standard library. That + might be because it would make race conditions very easy to trigger. Imagine checking to see if two + <code>LockCell</code> values were equal for an if-block, but once you get inside the if-block, + they're no longer equal. I'll have to think more about this. +</p> + +<pre> +let a: LockCell<i32>; +let b: LockCell<i32>; + +if a == b { + // a and b are no longer equal +} +</pre> + +<h3>Readonly Lock Collections</h3> + +<p> + I originally had a different idea of what to do with the <code>Sharable</code> trait. We could avoid + checking for duplicates in a lock collection if we knew that the only thing we would do with them is + read. At the time, I thought that reading twice on the same thread will never cause a deadlock. + However, that's not true on every platform, because of contention. Luckily, the <code>lock_api</code> + developers thought ahead by designing a <code>RawRwLockRecursive</code> trait, which I can extend by + creating a <code>Recursive</code> subtrait of <code>Lockable</code>. +</p> + +<p> + That's cool, but I don't really want to add three more lock collection types for this feature. So + maybe this could be a wrapper around the existing lock collection types, i.e. + <code>Readonly<<wbr/>RetryingLockCollection<<wbr/>L>></code>. I'd need to implement a + <code>LockCollection</code> trait that would have an associated unsafe constructor for creating the + collection without checking for duplicates. Then, creating that retrying lock collection would be + free (but locking it can still sometimes be expensive). +</p> + +<p> + This doesn't only apply to <code>RwLock</code>. If there was a <code>ReentrantMutex</code> type, then + we could implement <code>Recursive</code> on that, and use that in a readonly lock collection as well. + For those of you who don't know, a <code>ReentrantMutex</code> can be locked several times on the same + thread, but not while it's being locked by a different thread. However, the guard can only access the + data immutably, because multiple mutable references to data on the same thread is still undefined + behavior. It's mostly intended for wrapping <code>Cell</code> or <code>RefCell</code>. Unfortunately, + <code>lock_api</code> doesn't have a trait for recursive mutexes. Instead it provides a struct + wrapper for <code>RawMutex</code>, called <code>RawReentrantMutex</code>. This is despite the fact + that Unix's pthread API already supports recursive mutexes. I might be able to sway the + <code>lock_api</code> developers to support this. +</p> + +<h3>No Standard Library</h3> + +<p> + Doing this without the standard library would be very interesting. But currently, the + <code>Lockable</code> interface requires a <code>Vec</code> of pointers, so it might not be feasible. + The thread keys are stored in a thread local storage too, but I'll probably come up with some + <code>UnsafeKey</code> system to deal with that eventually, in order to support async runtimes. +</p> + +<h3>Other Types from the Standard Library</h3> + +<p> + It would be fun to add <code>OnceLock</code> or <code>LazyLock</code>, but I don't think it's + necessary. I don't think anybody has ever accidentally deadlocked using those interfaces, if at all. + In the past I've also considered <code>Condvar</code> and <code>Barrier</code>, but that's trickier. + They're kind of the opposite of a mutex, so tricks that work for avoiding deadlocks here won't work + on them. I've also never used either of those types before, so I'm probably not the best person to + figure that out. +</p> + +<h2>Conclusion</h2> + +<p> + This has been a very long nerd snipe. But I think this idea could become incredibly useful, which is + why I keep pushing for it. It'd be even better if it was a part of the standard libary to begin with, + but I'm aware that's not possible anymore. +</p> + +<p> + The fact that this is possible makes me appreciate the borrow checker even more. People often think + of it as a limitation (myself included), but it solves so many problems that I'm surprised more + languages don't adopt it. Not only does it solve memory corruption, but also resource leaks, and now + deadlocks. You spend more time specifying your code, but less time debugging it. +</p> + +<p> + There's still work that needs to be done (I haven't even begun async runtime support yet), but I hope + that as this library becomes more developed, there will be more projects which are immune this error. +</p> + +</article> +</body> diff --git a/src/blog/linking.kuht b/src/blog/linking.kuht new file mode 100644 index 0000000..f994765 --- /dev/null +++ b/src/blog/linking.kuht @@ -0,0 +1,245 @@ +<import "base.kuht" as "base" /> + +<head> + <title>What Other Distros Can Learn from NixOS</title> + <meta name="description" content="Traditional distros will either have delayed updates or unstable updates. NixOS allows us to have both." /> +</head> + +<body> +<article> +<h1>What Other Distros Can Learn from NixOS</h1> + +<p> + As a warning, this blog post contains the ramblings of an Arch user who + barely knows what a flatpak is and has like, three of them on their system. + Their background is in web deployment pipelines and they're trying to apply + their personal beliefs to Linux distributions, like an idiot, without even + making a proof-of-concept. Reader discretion is advised. +</p> + +<p> + One debate that often comes up is static versus dynamic linking. Most Linux + distributions try to manage dependencies by dynamically linking them. This can + be good, because if a previous version of a library contains a bug, then it's + easy to distribute the bug fix and have it apply to all applications + simultaneously. The downside is that if a newer version of a library contains + a bug, or a backwards-incompatible change, then that will apply to all + installed applications. Dynamic linking is a double-edged sword that makes it + both easier to fix bugs and easier to create new ones. +</p> + +<p> + But NixOS offers strict control over both the release and use of libraries, + which allows applications to get both the ease of rollout that comes with + dynamic linking, and the control that comes from static linking. +</p> + +<h2>Background</h2> + +<p> + When an application is installed on NixOS, it is not put in the + <code>/bin</code> folder. In fact, this folder is almost entirely empty. + Instead, it goes to <code>/nix/store</code>, in a directory with an + unreadable hash. Then, when you boot the system, the files for your system + applications are symlinked into <code>/run/current-system/sw</code>, which is + in the <code>PATH</code>, so these applications can run in your terminal. +</p> + +<p> + But that's only system applications. For libraries, the application sets its + own <code>LD_LIBRARY_PATH</code> variable. This tells the dynamic linker + where to fetch the required libraries from. This isn't exactly what the file + looks like, but you can imagine it looking something like this: +</p> + +<pre> +#!/path/to/bash +LD_LIBRARY_PATH="$LD_LIBRARY_PATH:/nix/store/naioghuraopjfi4738-libgit2-2.23" +export LD_LIBRARY_PATH +exec -a git "$@" +</pre> + +<p> + The benefit of this system is that you can multiple versions of the same + library installed on your system, and different applications can depend on + different versions, depending on what they need. +</p> + +<p> + I want to note that there's nothing special about NixOS that would make this + impossible on other systems. Flatpak already allows applications to bundle + their own versions of libraries, and Snap has a similar system. So I think + other distros can still benefit from this system. +</p> + +<h2>Testing</h2> + +<p> + Before I go into the benefits of the system described above, I'll touch on + something that distros can do already, but that specifying dependency + versions might make easier. +</p> + +<p> + Before you can even update the repository, you'll have to test the library. + Hopefully, the maintainer of the library has already done this for you, by + creating a suite of unit tests that they ran before creating the release. I + suppose it wouldn't hurt to double check. +</p> + +<p> + The more interesting test you'll need to run is similar to what + <a href="https://github.com/rust-lang/crater">crater</a> does for the Rust + ecosystem. Whenever a change is made to the language, Rust runs crater to + determine if the change would cause a breaking change for any packages on + crates.io. It's a very brute-force way to test for new bugs. +</p> + +<p> + In order for this to work, not only does the library need a unit test suite, + but so does every application that depends on that library. You'll build + every single dependent against the newer version of the library, and then + test all of them to see if anything breaks. +</p> + +<p> + Inevitably, some applications will fail under a new version of a library. + This doesn't necessarily need to be because of a problem in the library + itself, but it could be due to a transient failure in the application. In + such a scenario, the application's dependency would be updated to say + <code>3.13</code> instead of <code>3</code>, so it doesn't block other + applications from using the newer version. +</p> + +<p> + It would also be beneficial to store somewhere in the package which version + of each dependency an application was last tested with. Then, if you want to + be extra safe, you can delay upgrading the dependency until you know that a + test has been run on the latest dependencies. +</p> + +<h2>Beta Channel</h2> + +<p> + The delayed release of packages is what separates Arch Linux from Manajaro. + Some users are more interested in being on the bleeding edge than others. I + don't see any reason why two separate package repositories are needed for + this. We can allow users to opt-in to the bleeding edge for specific packages + faster. These people can submit bug reports before the new features make + their way to regular users. +</p> + +<p> + NixOS has this in the form of the <code>unstable</code> channel. Users can + opt-in to using the unstable channel to get an experience more similar to a + rolling release distro. And judging from some Reddit threads, there are + people using the unstable channel regularly. This makes sense to me, since + not everyone who uses Nix is there for stability. Some people just like + having a central config file. NixOS also lets people go one step further and + use <code>unstable</code> for specific packages, but leave the rest of their + entire system stable. +</p> + +<p> + Other distros do have this. Most commonly this comes in the form of an LTS + variant of the distro, which most users are expected to stay on. The non-LTS + variant is often treated as a beta. +</p> + +<h2>Rollout</h2> + +<p> + The rest of the ideas presented in this post are ideas that can be done as + long as you have a way to pin dependency versions, but they're not done by + NixOS. How important these are is debatable, since NixOS already tests their + stable channels before releasing them. +</p> + +<p> + Not every package is of equal importance. A change to a library affecting + pacman is far more critical to a library that only affects GIMP. But thanks + to the system we have here, it doesn't need to affect every package at the + same time. We can rollout a library change to less important packages first + before we apply it to every library. +</p> + +<p> + To give an example of how this would work, let's say a new library version + releases, and it affects both pacman and GIMP. For this to work, we'd need to + assign a priority level to every package. Pacman would probably be a level 2 + package, meaning it's very important. There should only be a few level 1 + packages. GIMP would be level 4, so not very important. As a level 4 package, + GIMP would be one of the first packages to use the new version of the + library. After a couple days, pacman would also start using the new library. +</p> + +<h2>Rollback</h2> + +<p> + So with all that in mind, what happens when a library actually is bad? Well, + marking the library version as bad is luckily pretty simple. We just update + the version resolver to say that if a package asks for version + <code>3</code>, don't use <code>3.14</code>. Prefer <code>3.13</code> + instead. Pacman already caches older versions of packages, so you might not + even need to re-install <code>3.13</code>. +</p> + +<p> + Of course, rolling backwards isn't always safe. Some newer packaages might + already be using features of <code>3.14</code> that don't exist in + <code>3.13</code>. To partially resolve this, we can run the test suite + against each package before downgrading to <code>3.13</code>. Hopefully + everything will pass, but if a particular package does not, then we can set + that package specifically to require a minimum of <code>3.14</code>. We can + also mark this downstream package as bad, and prevent new installations of it + until a patch is made. +</p> + +<h2>Distribution</h2> + +<p> + There's one thing here that we can't control: the distribution. Meaning, we + don't know when the user is going to run an upgrade. Unlike in Windows land + or web land, we don't run the upgrade automatically. Some systems will at + least automatically check for updates and notify the user when there is one, + but there's usually no distinction between critical bug fixes and feature + updates. +</p> + +<p> + I would recommend having a daemon that at least automatically downgrades in + the case of bad library versions. Since we can cache older versions of the + library, this wouldn't require too much network activity. And we can always + give the user an option to disable it. +</p> + +<p> + NixOS actually does support full automatic upgrades in the background. Some + NixOS users report that because NixOS is so stable, they feel comfortable to + just letting everything update in the background without intervention. The + NixOS wiki provides this example on how to do so. +</p> + +<pre> +system.autoUpgrade = { + enable = true; + flags = [ + "--print-build-logs" + ]; + dates = "02:00"; + randomizedDelaySec = "45min"; + allowReboot = false; # Set to true if you want automatic reboots +}; +</pre> + +<h2>Conclusion</h2> + +<p> + I don't want to say none of these solutions come with caveats. NixOS has + trouble running arbitrary executables because of how it goes about + accomplishing its goals. The way things are set up on NixOS can sometimes be + confusing to newcomers, and I understand why a Linux distro wouldn't want to + copy everything from NixOS. But I think even a best-effort attempt at some of + the ideas that NixOS provides could provide major benefits for stability. + It's just about choosing the right compromises. +</p> diff --git a/src/blog/named-optional-args.kuht b/src/blog/named-optional-args.kuht new file mode 100644 index 0000000..7d36972 --- /dev/null +++ b/src/blog/named-optional-args.kuht @@ -0,0 +1,1701 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Named and Optional Arguments are Awesome</title> + <meta name="description" content="Whenever someone asks what my least favorite part of Rust is, my answer is always the same: it doesn't have named or optional arguments. This article explores the ideas for adding them, both good and bad." /> +</head> + +<body> + +<article> + +<h1>Named and Optional Arguments are Awesome</h1> + +<p> + Whenever someone asks what my least favorite part of Rust is, my answer is + always the same: it doesn't have named or optional arguments. +</p> + +<p> + I have a small list of programming languages I tolerate: Rust, C#, TypeScript, + Dart, Python. Of those languages, Rust is the hardest to get named arguments + out of. I'm going to start this post by going over the other languages and how + they've achieved named arguments. Then, we'll look at Rust, the problems that + have resulted from not having named arguments, and proposals for how Rust + could get them in the future. +</p> + +<h2>How Other Languages Handle Named Parameters</h2> + +<h3>Dart</h3> + +<p> + Dart is designed specifically for graphical user interfaces. GUI components + tend to have lots of optional parameters, so Dart tried very hard to get them + right. And I think it does the best job out of any language we're going to + talk about today. +</p> + +<p> + First, let's look at a normal function, just to familiarize ourselves with the + language: +</p> + +<pre> +Widget Text(String text) { + // ... +} +</pre> + +<p> + That's pretty normal as far as languages go. You can probably guess what it + does, even if you don't know Dart. Dart was basically designed in a lab to be + easy for programmers to learn. +</p> + +<p> + Now, let's add a named parameter to set the size of the text, and default it to + 16 pixels: +</p> + +<pre> +Widget Text(String text, {int fontSize = 16}) { /* ... */ } + +// usage: +Text("Hello, world!", fontSize: 24) +</pre> + +<p> + Named parameters are wrapped in curly braces. This makes them look similar to + a map. Named parameters can also be defaulted to null, or even required. +</p> + +<pre> +Widget Text({ + int fontSize = 16, + Color? color, + required String text, +}); +</pre> + +<p> + Dart also supports optional positional arguments by wrapping the parameter in + square brackets. +</p> + +<pre> +// From the Dart documentation +String say(String from, String msg, [String device = 'carrier pigeon']) { + var result = '$from says $msg with a $device'; + return result; +} + +assert(say('Bob', 'Howdy') == 'Bob says Howdy with a carrier pigeon'); +assert(say('Bob', 'Howdy', 'smoke signal') == 'Bob says Howdy with a smoke signal'); +</pre> + +<h3>C#</h3> + +<p> + C#'s handling of named parameters is also pretty good. In fact, any parameter can be + named. +</p> + +<pre> +void ExampleMethod(string foo, string bar) +{ + Console.WriteLine($"{foo} {bar}") +} + +ExampleMethod(bar: "Hello", foo: "world"); +</pre> + +<p> + And any parameter can have a default value, as long as they're specified after + the required parameters. +</p> + +<pre> +void ExampleMethod(string foo, string bar = "world") +{ + Console.WriteLine($"{foo} {bar}") +} + +ExampleMethod("Howdy"); +</pre> + +And that's all there is to it. I like the simplicity of it. + +<h3>TypeScript</h3> + +<p> + TypeScript's handling of named parameters is not very good, in my opinion. + But optional positional parameters aren't named, and they work pretty well, so + let's start with that. +</p> + +<pre> +function example(required: string, optional?: string = "world"): string { + return `${required} ${optional}`; +} + +example("Hello") === "Hello world"; +example("Hello", "TypeScript") === "Hello TypeScript"; +</pre> + +<p> + That seems fine to me. If you don't specify the default value, then it gets + set to <code>undefined</code>. The problem is that there isn't a good built-in + syntax for named parameters, so we have to use other TypeScript features to + hack it in. +</p> + +<pre> +function example( + required: string, + { + requiredNamed, + namedOptional = "world" + }: { requiredNamed: string, namedOptional?: string } +): string { + return `${required} ${requiredNamed} ${namedOptional}`; +} + +example("Hello", { requiredNamed: "beautiful" }); +</pre> + +<p> + Here, we're taking advantage of both anonymous types and destructuring. The + second parameter of the function has an anonymous object type, which contains + two fields: <code>requiredNamed</code> and <code>namedOptional</code>. Then + we destructure this parameter so we can use the fields in the function. +</p> + +<p> + The most annoying part of this is that we have to define the field names + twice. Once for the type, and once for the function parameters. This is + unnecessary verbosity that any good language should try to avoid. The function + call is also more verbose than necessary, because of the curly braces. +</p> + +<p> + Unfortunately, this syntax is used all of the time in React. Components are + typically a function that takes one argument, which is an object. So when you + define a component, you have to do this constantly. +</p> + +<h3>Rust</h3> + +<p> + Some people have tried to do named parameters in Rust. This generally doesn't + work very well. The most common approach is to define a struct for the + function, and then have it implement the <code>Default</code> trait, so that + we can end the struct initializer with <code>..Default::default()</code>. +</p> + +<pre> +struct ExampleParams { + foo: String, + bar: String, +} + +impl Default for ExampleParams { + fn default() -> Self { + Self { + foo: "Hello".into(), + bar: "Hello".into(), + } + } +} + +fn example(Example { foo, bar }) -> String { + format!("{foo} {bar}") +} + +assert_eq!( + example(Example { foo: "Hello".into(), ..Default::default() }), + "Hello world".to_string() +); +</pre> + +<p> + I complained before about how verbose TypeScript's version is, but this is + even worse, because the types are not anonymous. And worse, if your default + values are not the default of the types (e.g. empty string for <code>String</code>), + then you need to write a whole function to set the default values. It also + only works if every single parameter has a default value. If only some of the + parameters have default values, then you need an even more verbose solution. +</p> + +<h2>Why Rust Needs Optional Parameters</h2> + +<p> + The standard library doesn't tend to follow the above pattern (for good reason, + in my opinion). But there are still several cases where the standard library + probably would have benefitted from it. +</p> + +<p>Take some of the factories of <code>HashMap</code>, for example.</p> + +<pre> +impl<K, V> HashMap<K, V, RandomState> { + pub fn new() -> HashMap<K, V, RandomState>; + pub fn with_capacity(capacity: usize) -> HashMap<K, V, RandomState>; +} + +impl<K, V, A: Allocator> HashMap<K, V, RandomState, A> { + pub fn new_in(alloc: A) -> Self; + pub fn with_capacity_in(capacity: usize, alloc: A) -> Self; +} + +impl<K, V, S> HashMap<K, V, S> { + pub const fn with_hasher(hash_builder: S) -> HashMap<K, V, S>; + pub fn with_capacity_and_hasher(capacity: usize, hasher: S) -> HashMap<K, V, S>; +} + +impl<K, V, S, A: Allocator> HashMap<K, V, S, A> { + pub fn with_hasher_in(hash_builder: S, alloc: A) -> Self; + pub fn with_capacity_and_hasher_in(capacity: usize, hasher: S, alloc: A) -> Self; +} +</pre> + +<p> + With just three optional parameters, we need eight functions to represent all + of the overloads. Each one of them needs to have their own name, since Rust + doesn't actually have overloads. And the programmer and/or code reviewer must + either memorize or look up the order of the parameters when one of the more + complex factories are used. +</p> + +<p> + Another problem that comes up is the fact that, when the parameters are + unnamed, it can be easier to make mistakes. Consider the following print + function. +</p> + +<pre> +fn print(text: &str, bold: bool, italics: bool, underline: bool); +</pre> + +<p> + It's easy to mix up the parameters here because they don't have names. If I + forget that the <code>italics</code> parameter is third and not second, then I + can end up accidentally bolding my text instead. +</p> + +<p> + Some functions just have lots of parameters as well. Imagine if I added color, + underline color, background color, blinking text, hidden text, circled text, + and fast blinking text above. Nobody wants to pass in 14 arguments to a + function, of which several will have sensible defaults. And I don't see much + reason to name a new struct if it's only going to be used for the one function. +</p> + +<h2>Proposals for Named Arguments</h2> + +<p> + First I'm going to describe some proposals that I've seen for improving named + arguments in Rust, that I don't like. Then I'll reveal what my preferred + solution is. +</p> + +<h3>Default Field Values</h3> + +<p> + This feature is already available in Nightly Rust. And the <code>syn</code> + create even recently added support for it. The idea is to make implementing + the <code>Default</code> trait easier by adding special syntax for it. It's + not proposed that this will solve named arguments by itself, but might be part + of a larger solution. +</p> + +<pre> +#[derive(Default)] +struct Pet { + name: Option<String>, // impl Default for Pet will use Default::default() for name + age: i128 = 42, // impl Default for Pet will use the literal 42 for age +} + +// Pet { name: Some(""), age: 42 } +let _ = Pet { name: Some(String::new()), .. }; +// Compilation error: `name` needs to be specified +let _ = Pet { .. }; +// Pet { name: None, age: 42 } +let _ = Pet::default(); +</pre> + +<p> + I don't actually have a problem with the proposal itself. I even created my + own library to implement a polyfill for it + (<a href="https://www.lib.rs/feluments">feluments</a>). It seems perfect for + structs. But it doesn't help very much for functions. +</p> + +<h3>Structural Records</h3> + +<p> + The RFC for default field values mentions a closed RFC for structural records. + They can be thought of as anonymous structs or tuples with named fields. +</p> + +<pre> +fn do_stuff_with(color: { red: u8, green: u8, blue: u8 }) { // More ergonomic! + some_stuff(color.red); // *And* readable! :) + ... + other_stuff(color.green); + ... + yet_more_stuff(color.blue); +} + +do_stuff_with({ red: 255, green: 127, blue: 63 }); +</pre> + +<p> + The RFC notes that this emulates named arguments, although it does not emulate + optional arguments. It wasn't a major motivation. The major motivation was to + get the convenience of tuples with the readability of structs. +</p> + +<p> + Let's say for the sake of argument that this RFC succeeded. What would the + syntax of our "Hello, world" examples look like here? Let's assume default + field values are also merged, and they work with structural records. +</p> + +<pre> +fn example( + required: String, + {requiredNamed, namedOptional}: { + requiredNamed: Option<String>, + namedOptional: String = "world".into() + } +): String { + format!("{required} {requiredNamed} {namedOptional}") +} + +example("Hello", { requiredNamed: "beautiful", .. }); +</pre> + +<p> + Wow, that looks familiar. In fact it's the same syntax that TypeScript uses, + with all the problems that come with it. You need to specify the parameter + names twice. +</p> + +<p>Structural records also come with many other challenges:</p> +<ul> + <li> + What happens with records that have only one field? <code>{ x }</code> can be + interpreted as either a block expression that returns <code>x</code>, or a + structural record expression: <code>{ x: x, }</code>. To fix the ambiguity, + the RFC proposed requiring an extra comma: <code>{ x, }</code>, like what + tuples do. But this requires making the Rust parser much more complicated, + to do several tokens of lookahead. + </li> + <li> + It's impossible to implement any traits on these. The RFC proposed magically + implementing several traits. And unlike for tuples, we can't just define some + generic trait implementations, because there can be so much variation in the + field names. + </li> + <li> + A newcomer to the language may mistake this syntax for a Python dictionary + or Dart's map type. + </li> +</ul> + +<p> + Interestingly, TypeScript actually has named tuple fields. But they only exist + at the type declaration. You can't use the field names to access the fields. +</p> + +<pre> +type NewLocation = [lat: number, long: number] + +const newLocations: NewLocation[] = [ + [52.3702, 4.8952], + [53.3498, -6.2603] +] + +const firstLat = newLocations[0][0] +const firstLong = newLocations[0][1] +</pre> + +<p> + Rust's language team decided that although something resembling this feature + seems useful, it would be very hard to implement and not worth the effort. +</p> + +<h3>Copy C#</h3> + +<p> + If we're already agreeing on a syntax for optional fields in structs, why not + just copy that over to functions? First, we can allow the arguments to be + specified in any order, as long as they are named, and no out-of-order + positional arguments are following a named argument. +</p> + +<pre> +print_order_details(order_num: 31, product_name: "Red Mug", seller_name: "Gift Shop"); +print_order_details(seller_name: "Gift Shop", product_name: "Red Mug", order_num: 31); +print_order_details("Gift Shop", 31, product_name: "Red Mug"); +print_order_details(seller_name: "Gift Shop", 31, product_name: "Red Mug"); + +// this would cause an error +print_order_details(product_name: "Red Mug", 31, "Gift Shop"); +</pre> + +<p> + Then, we could define default values, and allow parameters with default values + to be unspecified. +</p> + +<pre> +fn print_order_details(product_name: &str, order_num: 31, seller_name: &str = "Gift Shop"); + +print_order_details("Red Mug", 31); +</pre> + +<p> + Unlike for the default field values proposal, we can't use a <code>..</code> + at the end here, because it is also a valid expression. +</p> + +<p> + The biggest problem with this idea is that it makes parameter names part of + the public API of the function, without functions opting into it. There was + previously an objection on the grounds of it conflicting with the proposed + type ascription feature. But the RFC for it has since been deleted, so I don't + think we need to worry about that anymore. +</p> + +<h3>Public Arguments</h3> + +<p> + As far as I know, a formal RFC was never made for this, but a pre-RFC was + posted to the Rust Internals Forum several years ago, and it's the most + thorough proposal for named arguments that I know of. I'll avoid copy-pasting + the entire RFC here, but I'll give a high level overview. +</p> + +<p>Here's an example of a function with named arguments from the RFC:</p> + +<pre> +pub struct Database; +pub struct RegistrationError; + +pub fn register( + pub name: String, + pub surname: String, + to db: Database +) -> Result<(), RegistrationError> { + /* ... */ +} + +register(name: "Alexis".into(), surname: "Poliorcetics".into(), to: my_db); +</pre> + +<p> + In the case of <code>to db</code>, the name of the argument is <code>to</code>, + but it's referred to as <code>db</code> within the body of the function. For + the other two parameters, the <code>pub</code> keyword is used to say that + that both the name of the argument when calling the function, and the name of + the parameter in the function body, are the same. +</p> + +<p> + The RFC goes into way more detail, proposing function overloads, using named + arguments with the <code>Fn</code> trait, exposing names from patterns, method + overloads, and explaining the documentation aspect. +</p> + +<p>Here are the criticisms I saw in the comments:</p> +<ul> + <li> + Unordered named arguments was ruled out. I don't like the justification + RustyFrog gave for this, but it seems easy to support with this model. + </li> + <li> + Default parameter values also weren't proposed, even though it's the main + reason for wanting named parameters + </li> + <li> + Despite not proposing default parameter values, the pre-RFC did propose + overloads, which are very difficult to implement in Rust, and they're not as + useful as optional arguments. It's not proposing type-based overloads, which + is a bit better, but still. + </li> + <li>It's a very big RFC, which might make it difficult to get it approved.</li> + <li> + There are some syntax ambiguities, like <code>fn foo(name (a, b): name)</code>, + which is currently valid syntax. There was also a proposed type acription + feature that it conflicted with, but as I said before, that RFC has since + been removed. + </li> + <li> + Using <code>pub</code> might be confusing, given how it's used currently. I + don't think I mind that personally, because its effect is to make the name of + the parameter part of a function's public API. But I get how that's a bit + weird. + </li> + <li> + The proposal for changing the <code>Fn</code> trait seemed underbaked. The + <code>Fn(f32, f32) -> String</code> trait gets desugared to + <code>Fn<(f32, f32), Output = String></code>, but it's not clear what the + desugaring for named parameters should look like. + </li> + <li> + Despite explaining the usage for <code>Fn</code> implementations, the <code>fn</code> + type was not mentioned at all. + </li> + <li> + The pre-RFC was not clear about if named parameters were required to be used + by name, or if having the correct position was technically sufficient. + </li> + <li> + The handling of extern functions with named arguments is fairly complicated + to the point where it would probably be better to not handle them. + </li> +</ul> + +<p> + My understanding is that the author got busy and didn't address these concerns, + but I think the pre-RFC serves as a good starting point. +</p> + +<h3>Struct-style function calls</h3> + +<p> + This idea was proposed back in 2016 by nixpulvis. I've also seen it floating + around in some other places. +</p> + +<pre> +// 1. Tuple style method call (current functions). +fn foo(a: u32) {} +foo(2); + +// 2. Struct style method call. +fn foo { a: u32 } +foo { a: 2 }; +// SomeType { a: 1 } -> SomeType +// SomeFunc { a: 1 } -> Codomain +// generally_lowercase { a: 1 } -> Codomain + +// 3. Together. +fn foo(a: u32) { a: u32 } {} +foo(1) { a: 2 }; +</pre> + +<p> + I personally don't like it because it makes struct constructions and function + calls look too similar. I also have no idea how the compiler would parse this. + But then again, most languages use <code>new Foo()</code> syntax for + constructors, so maybe somebody will like it. +</p> + +<h3>Copy Dart</h3> + +<p> + Another proposal that's come up, including from myself, is to copy the Dart + syntax, like so: +</p> + +<pre> +fn foo({ foo: &str = "Hello", bar: &str = "World" }) -> String { + format!("{foo} {bar}") +} +</pre> + +<p> + To me, this looks too much like existing pattern syntax. But I do like that it + doesn't require an extra keyword on each argument. +</p> + +<h3>Edition Separation</h3> + +<p> + Some people have proposed solving the backwards-compatibility problem by only + allowing named arguments to be used with functions that were written in a + future edition of Rust. To me, this feels like too large of a breaking change + to be worth considering. +</p> + +<h3>Dot Prefixes</h3> + +<p> + An RFC was opened in 2020 proposing using a dot prefix to distinguish between + named and unnamed variables. It was not merged, because the RFC was a very + incremental step that didn't explain the future plans for how the named + arguments would work. Here's an example from the PR: +</p> + +<pre> +fn split(string: &str, .at: char, .limit: usize, .case_sensitive: bool) {} + +split("hello world", .at = ' ', .limit = 2, .case_sensitive = true); +</pre> + +<p> + The dot was chosen because it's only one character, (as opposed to four + characters, with the <code>pub</code> keyword plus a space). But I honestly + don't think it's intuitive enough. The dot is also reminds the user of struct + fields, which is a completely unrelated concept. +</p> + +<h2>My Proposal</h2> + +<p> + After reading all of these proposals, I feel that all of the pieces exist for + a complete named arguments proposal that lack any of the downsides. I get the + impression that some people are going to be opposed to the concept no matter + what, but I tried to address as many counter-arguments for named arguments as + possible in this post. The rest of this post will be a pre-RFC that think + will satisfy as many people as possible. I'll monitor the comments anywhere + this gets posted to see if there are any major criticisms. I can't guarantee + that I'll actually make an RFC, but permission is granted to create an RFC if + I don't. I think it's been a while since named argument were last seriously + proposed, so I think now is a good time for a new proposal. +</p> + +<h3>Summary</h3> + +<p> + Add named and optional arguments to functions. Functions can have both + positional and named arguments. In function calls, named arguments may be + referred to by name, to increase readability and maintainability. Named + arguments may also have default values, avoiding the need to specify them in + the function call. +</p> + +<h3>Motivation</h3> + +<h4>Improve safety and readability</h4> + +<p> + Misremembering which parameter is at which position is a major source of bugs + in many languages, including Rust. Take the function from the standard + library: <code><a href="https://doc.rust-lang.org/stable/std/fs/fn.hard_link.html">hard_link</a></code>. +</p> + +<pre> +pub fn hard_link<P: AsRef<Path>, Q: AsRef<Path>>( + original: P, + link: Q, +) -> Result<()> +</pre> + +<p> + Both parameters accept the same types, so there's no way to know which one is + which without looking at the documentation. The consequence of misremembering + is a runtime error or a bug. +</p> + +<p> + With named parameters, it is easy to see what is happening at the call-site. + This also makes the resulting code more readable. +</p> + +<pre> +hard_link(original: "a.txt", link: "b.txt")?; +</pre> + +<p>The effect is more pronounced for functions which have more parameters.</p> + +<pre> +// Without named arguments: what are these numbers? +solar_elevation(1787517098.0, 38.897957, -77.036560) + +// With named arguments, it's obvious +solar_elevation(timestamp: 1787517098.0, latitude: 38.897957, longitude: -77.036560); +</pre> + +<p> + A similar effect is possible in Rust today, by creating a new struct and + having that be the function's only parameter. But it's very verbose to create, + so in practice, few libraries take advantage of it. Importantly, the standard + library rarely uses this pattern. +</p> + +<h4>Boilerplate reduction</h4> + +<p> + Many functions have reasonable defaults that should be passed into many of + their parameters, but because all positional arguments are required, all of + the parameters must be specified. Passing <code>None</code> into every + parameter comes with readability issues. +</p> + +<pre> +repository.checkout_index(None, None)?; +</pre> + +<p> + Passing a struct into the parameters does give us names, but still requires + us to specify <code>None</code> for the optional values. Sometimes this + boilerplate can be reduced by adding <code>..Default::default()</code> to the + end of the constructor, <code>Default</code> cannot be implemented if any of + the fields are not optional. Implementing <code>Default</code> is also a lot + of boilerplate if it cannot be implemented simply by using the derive macro. + Some of these problems can be addressed by + <a href="https://github.com/rust-lang/rfcs/pull/3681">RFC 3681</a>, but it + still requires a new struct to be defined specifically for one function. +</p> + +<pre> +let instance = Instance::new(InstanceDescriptor { + backends: Backends::default(), + flags: InstanceFlags::default(), + memory_budget_thresholds: MemoryBudgetThresholds::default(), + backend_options: BackendOptions::default(), + display: None, +}); +</pre> + +<p> + To help reduce the boilerplate of calling such a function, many libraries + implement builder types that don't require every value to be specified. This + comes at the cost of even more boilerplate to create the builder, which + becomes part of the library's public API. There are crates, such as + <a href="https://www.crates.io/crates/bon">bon</a> to make declaring a builder + for a struct easier, but it still requires a struct be declared for a single + function. Libraries don't always use these crates either, as doing so + increases compile times. The <code>Command</code> struct uses 60 SLOCs (and + many more lines of documentation) to allow us to do this: +</p> + +<pre> +Command::new("printenv") + .stdin(Stdio::null()) + .stdout(Stdio::inherit()) + .env_clear() + .envs(&filtered_env) + .spawn() + .expect("printenv failed to start"); +</pre> + +<p>If we had named arguments, the following function would be easy to define:</p> + +<pre> +spawn( + command: "printenv", + stdin: Stdio::null(), + stdout: Stdio::inherit(), + env_clear: true, + envs: &filtered_env +).expect("printenv failed to start"); +</pre + +<p> + Bon also provides utilities for emulating named arguments in functions, but + calling such a function involves some boilerplate as well. +</p> + +<pre> +// Example from the bon README + +#[builder] +fn greet(name: &str, level: Option<u32>) -> String { + let level = level.unwrap_or(0); + + format!("Hello {name}! Your level is {level}") +} + +// This could be shortened to greet(name: "Bon", level: 24) +let greeting = greet() + .name("Bon") + .level(24) + .call(); + +assert_eq!(greeting, "Hello Bon! Your level is 24"); +</pre> + +<p> + In practice, the added complexity means that these patterns are only ever used + in public APIs, so private functions have no worthwhile workaround to improve + readability. +</p> + +<h4>Extensibility without breaking changes</h4> + +<p> + Currently, adding a new parameter to an existing function is a breaking + change. If you want to add a new field to a struct, your existing factory must + implement a reasonable default value, and a new function must be created to + enable the new functionality. +</p> + +<p> + This leads to some APIs having many functions which do similar things, but + with different parameters. For example, <code>HashMap</code> from the standard + library has the following factories. +</p> + +<pre> +impl<K, V> HashMap<K, V, RandomState> { + pub fn new() -> HashMap<K, V, RandomState>; + pub fn with_capacity(capacity: usize) -> HashMap<K, V, RandomState>; +} + +impl<K, V, A: Allocator> HashMap<K, V, RandomState, A> { + pub fn new_in(alloc: A) -> Self; + pub fn with_capacity_in(capacity: usize, alloc: A) -> Self; +} + +impl<K, V, S> HashMap<K, V, S> { + pub const fn with_hasher(hash_builder: S) -> HashMap<K, V, S>; + pub fn with_capacity_and_hasher(capacity: usize, hasher: S) -> HashMap<K, V, S>; +} + +impl<K, V, S, A: Allocator> HashMap<K, V, S, A> { + pub fn with_hasher_in(hash_builder: S, alloc: A) -> Self; + pub fn with_capacity_and_hasher_in(capacity: usize, hasher: S, alloc: A) -> Self; +} +</pre> + +<p> + All of these functions are shown in the documentation, polluting the API. The + number of functions is exponential with respect to the number of parameters. + If we ignore the differences in generic parameters (a possible solution to + which is discussed in the "Future possibilities" section), then we could + imagine having a single function. +</p> + +<pre> +pub fn new( + pub capacity: usize = 0, + pub alloc: A = Global, + pub hasher: S = RandomState, +); +</pre> + +<p> + If we wanted to some day add a <code>fill</code> parameter, we'd be able to + add it to this existing function without it being a breaking change, and + without doubling the number of factories for <code>HashMap</code>. +</p> + +<h3>Guide-level explanation</h3> + +<p> + Parameters may be named or unnamed. By default, all parameters to a function + are unnamed, and cannot be referred to by name. We can create a named + parameter by adding the <code>pub</code> keyword before its name. +</p> + +<pre> +fn print_labeled_measurement(pub value: i32, pub unit_label: char) { + println!("The measurement is: {value}{unit_label}"); +} + +fn main() { + print_labeled_measurement(5, 'h'); +} +</pre> + +<p> + At first, this may not seem very different from unnamed parameters. But when + a parameter is named, we may use its name when calling the function. This + makes it easier to tell, at a glance, what the values being passed into the + function are meant to do. +</p> + +<pre> +print_labeled_measurement(value: 5, unit_label: 'h'); +</pre> + +<p> + When parameters are named, they may also be re-ordered, so you don't need to + remember the position of each one. +</p> + +<pre> +print_labeled_measurement(unit_label: 'h', value: 5); +</pre> + +<p> + There are some caveats to the ordering. A function may contain both named and + unnamed parameters. But if it does, all named arguments must be specified + after the positional arguments, both in the function declaration and in the + function body. +</p> + +<pre> +// unit_label is named, value is not +fn print_labeled_measurement(value: i32, pub unit_label: char) { + println!("The measurement is: {value}{unit_label}"); +} + +// forbidden: all named parameters must be after all unnamed parameters +fn _print_labeled_measurement(pub value: i32, unit_label: char) { + // ... +} + +fn main() { + // forbidden: the parameter that 'h' is being passed into must be named + print_labeled_measurement(value: 5, 'h'); + // forbidden + print_labeled_measurement(unit_label: 'h', 5); +} +</pre> + +<p> + Named parameters may also have default values. This means that the caller does + not need to specify the argument value when calling the function. The default + value is using automatically. +</p> + +<pre> +fn print_labeled_measurement(pub value: i32, pub unit_label: char = 'm') { + println!("The measurement is: {value}{unit_label}"); +} + +fn main() { + // unit_label is unspecified and is defaulted to 'm' + print_labeled_measurement(5) +} +</pre> + +<p> + Named parameters cannot be used if the parameter does not have a name and is + defined as a pattern instead. For example, the following is not valid. +</p> + +<pre> +fn print_coordinates(pub &(x, y): &(i32, i32)) { + // ... +} +</pre> + +<p> + However, a name can be given to this parameter by using an <code>@</code> + binding. +</p> + +<pre> +fn print_coordinates(pub point @ &(x, y): &(i32, i32)) { + // ... +} +</pre> + +<p> + Traits may also choose to use named parameters, in which case the + implementation must use the same name. However, trait function parameters + cannot have default values. +</p> + +<pre> +trait Trait { + fn f(pub a: i32, pub b: i32); +} + +impl Trait for () { + // This is okay, because the parameter names match + // unused parameter warnings can be ignored using @ + fn f(pub a: i32, pub b @ _: i32) {} +} + +impl Trait for i32 { + // invalid: the second argument is named `b` in the definition + fn f(pub a: i32, pub c: i32) {} +} + +impl Trait for bool { + // invalid: `b` is not public + fn f(pub a: i32, b: i32) {} +} + +impl Trait for X { + // invalid: trait methods cannot have default parameter values + fn f(pub a: i32, pub b: i32 = 0) {} +} +</pre> + +<p> + Because the parameter names are optional, function pointers may be created for + functions with named parameters. However, the name will not be accessible on + the function pointer, and default values cannot be used. This also applies for + the <code>Fn*</code> family of traits. +</p> + +<pre> +fn foo(pub a: i32, pub b: i32) {} + +let f: fn(i32, i32) = foo; + +f(4, 2); // ok +f(a: 4, b: 2); // ERROR! `f` can't be called with named arguments + +fn higher_order(f: Fn(i32, i32)) { + f(4, 2); // ok + f(a: 4, b: 2); // ERROR! `f` can't be called with named arguments +} +higher_order(foo); // ok +higher_order(|_, _| {}); // ok +</pre> + +<h3>Reference-level explanation</h3> + +<h4>Grammar</h4> + +<p> + The only needed change to support public parameters in function declarations + is to function parameters. First, we add the <code>pub</code> keyword, which + must be followed by an identifier pattern. Such a pattern may also include a + default value. +</p> + +<pre> +FunctionParamPattern = PatternNoTopAlt ":" ( Type | "..." ) + | "pub" IdentifierPattern ":" ( Type | "..." ) ( "=" Expr )? +</pre> + +<p> + For function calls, the syntax is slightly more complicated to parse, but only + affects <code>CallParams</code>. +</p> + +<pre> +CallParam = Expression ( ":" Expression )? +CallParams = CallParam ( "," CallParam )* ","? +</pre> + +<p> + This is slightly more complicated to parse, because an identifier is also a + valid expression. To solve this, we'll just parse an entire expression, and + then check to see if there's a colon following it. The compiler can later + give an error if the argument name is anything other than an identifier. +</p> + +<h4>Static semantics</h4> + +<h5>Function parameters</h5> + +<p> + Given a <code>FunctionParamPattern</code> where the default is specified, i.e.: +</p> + +<pre> +FunctionParamPattern = "pub" pat:IdentifierPattern ":" ( ty:Type | "..." ) ( "=" expr:Expr ) +</pre> + +<p>similar rules to the rules for default field values apply. Namely,</p> +<ul> + <li> + If the function is <code>const</code>, then the expression, <code>expr</code> + must be a constant expression. + </li> + <li>The expression <code>expr</code> must coerce to the type <code>ty</code></li> + <li> + Generic parameters of the current items are accessible. +<pre> +fn foo<const A: usize>(bar: usize = A) {} +</pre> + </li> + <li> + Default const expressions are not evaluated at definition time, only during + instantiation. This means that the following will not fail to compile: +<pre> +fn foo(a: usize = panic!(), b: usize = 42) {} + +foo(a: 0); +</pre> + </li> + <li> + The function's generic parameters are properly propagated, meaning the following is possible: +<pre> +fn foo<T>(bar: Vec<T> = Vec::new()) {} + +foo::<T>(); +</pre> + </li> + <li> + When lints check attributes such as <code>#[allow(lint_name)]</code> are + placed on a <code>FunctionParam</code>, it also applies to the expression, + <code>expr</code>. + </li> + <li> + Unused parameter names will not emit a warning if they are bound to a + different pattern using <code>@</code>. The bound pattern may still emit a + warning + </li> +</ul> + +<p> + For public parameters in general, all public parameters must be strictly after + all non-public parameters. +</p> + +<h5>Function calls</h5> + +<p> + The semantics of function calls change significantly. In general, all named + arguments must be specified after all positional arguments. For positional + arguments, all existing semantics apply. +</p> + +<p>In the case of the following production of a named argument:</p> +<pre> +CallParam = name:Expression ":" value:Expression +</pre> +<p>the following rules apply</p> +<ul> + <li>The expression, <code>name</code> must be a plain identifier.</li> + <li> + The identifier, <code>name</code> must be the name of a public parameter of + the function. + </li> + <li> + The expression, <code>value</code> must coerce the the type of the + aforementioned public parameter. + </li> +</ul> + +<h5>Trait declarations</h5> + +<p> + For trait declarations all the same rules apply from function declarations. In + addition: +</p> +<ul> + <li>Trait methods cannot have default parameter values.</li> +</ul> + +<h5>Trait implementations</h5> + +<p> + The semantics of trait implementations vary slightly from the semantics for + trait declarations. +</p> +<ul> + <li> + Any parameter which was declared public on the trait declaration must also + be public in the trait implementation. + </li> + <li>Public parameters must have the same name that their declaration has.</li> +</ul> + +<h3>Drawbacks</h3> + +<h4>Added complexity</h4> + +<p> + The language would be slightly more complex as a result of this feature. But + this proposal is fairly minimal as far as named arguments proposals go. The + syntax for named arguments resembles the syntax for struct fields, and the + default value syntax is pretty similar to the syntax proposed for default + field values. +</p> + +<p> + One concern that can come up in named argument proposals is the problem of it + only applying to new functions, which makes it hard to remember which + libraries. This proposal can be backported to the standard library, making the + language feel more consistent. +</p> + +<h4>Named arguments aren't mandatory</h4> + +<p> + The fact that names are optional could be seen as a compromise on the safety + benefits. Some people may lazily choose to leave them out, in which case, no + safety benefit is provided. +</p> + +<p> + However, the fact that they aren't mandatory allows more library authors to + use them without worrying about breaking changes. Some authors may also be + more inclined to support named arguments if they felt confident that it would + not increase the verbosity of calling the function. +</p> + +<p> + This drawback does present a potential bug. If the names are removed during a + refactor, then they may be in the wrong position after the refactor, causing + a bug. +</p> + +<pre> +// before refactor +coordinates(y: 45.6, x: 6784.0); + +// after refactor: incorrect order +coordinates(y, x); +</pre> + +<h5>Remedy: Add a clippy lint</h5> + +<p> + A clippy lint can be used to enforce the use of named arguments when possible. +</p> + +<h5>Remedy: Require named arguments in a future edition</h5> + +<p> + Although it may be a large breaking change, future editions could likely + require the use of named parameters, if it were deemed desirable to do so. +</p> + +<h4>Limiting type ascription</h4> + +<p> + Many older proposals for named arguments noted that the syntax chosen here: + <code>name: value</code> conflicts with the proposed type ascription feature. + However, since then, the RFC for type ascription has been removed. And another + type ascription RFC was rejected, partially on the basis of it potentially + conflicting with named arguments. So this is not likely to be a concern + anymore. +</p> + +<h3>Rationale and alternatives</h3> + +<h4>Why make named arguments opt-in?</h4> + +<p> + Although many languages, such as Kotlin and C#, allow any argument to be + named, this is not practical for Rust. Changing an argument is currently not + a breaking change, but would become one as soon as named arguments are + introduced. The solution is to allow authors to decide if their argument names + are an implementation detail or not. +</p> + +<h4>The <code>pub</code> keyword</h4> + +<p> + Using the <code>pub</code> keyword may seem like a strange choice, given the + semantics. But it has essentially the effect of making the parameter name + public, so this is defensible. Other visibility modifiers are not proposed by + this RFC, because it's hard to imagine a situation where a public function + would want to allow named arguments internally, but not externally. +</p> + +<p> + Other modifiers have been proposed. One could imagine adding an <code>ext</code> + keyword to define named parameters. A previous RFC proposed using dots, but + this was deemed to be too reminiscent of field access. Other symbols, such as + <code>@</code> may be more acceptable. Another proposed idea is to copy Swift + syntax and use two identifiers. +</p> + +<pre> +fn foo(bar bar: &str) {} +</pre> + +<p> + Re-typing the same parameter name twice is verbose and usually unnecessary. + It's also unnecessary since syntax exists for rebinding a variable already. +</p> + +<p> + Dart wraps named arguments in curly braces. This has the benefit of not + requiring the <code>pub</code> keyword to be specified multiple times. However, + in the context of Rust, this looks like a destructuring pattern, which isn't + semantically correct. +</p> + +<pre> +// In Dart +void enableFlags({bool? bold, bool? hidden}) +enableFlags(bold: true) + +// Similar idea, in Rust +fn enable_flags({ bold: Option<bool> = None, hidden: Option<bool> = None }) +enable_flags(bold: Some(true)) +</pre> + +<p> + Another proposal is to make function calls with named arguments look more like + struct constructions by using curly braces instead of parentheses. This seems + like it could be very confusing and make the distinction between the two + features unclear. +</p> + +<pre> +fn foo (a: u32) { b: u32 } {} +foo(1) { b: 2 }; +</pre> + +<h4>Colon vs equal sign</h4> + +<p> + Some have proposed using an equal sign (<code>=</code>) instead of a colon + (<code>:</code>). This would make the syntax more closely resemble assignment, + which is semantically true. It also would not possibly conflict with type + ascription. +</p> + +<p> + The colon syntax was chosen for this RFC to be more consistent with struct + initializers. +</p> + +<h4>Argument re-ordering</h4> + +<p> + Some proposals for named arguments have omitted re-ordering of arguments for + simplicity. For the sake of thoroughness, this RFC proposes allowing arguments + to be re-ordered. A possible alternative would be to require arguments to + still be in the original order. The drawback would be unnecessary friction + when the user sorts the arguments out of order. +</p> + +<p> + Some languages allow for more flexible re-ordering than this RFC proposes. C# + allows named arguments to appear before unnamed arguments, as long as the + arguments are still in order. This RFC omits this feature, in order to make + named arguments seem less confusing to users. The reordering proposed here + still allows for more flexible argument-reordering in the future. +</p> + +<h4>No default values in traits</h4> + +<p> + This RFC currently prohibits traits from making use of default parameter + values. This is because it's unclear if default values for trait parameters + should be defined at the trait definition or the trait implementation. It + would likely be more predictable if default values could be defined by the + implementation. But it's hard to imagine how a default value would allow the + trait to be dyn-compatible. If the default values were defined at the trait + definition, then they could stay dyn-compatible. +</p> + +<h4>On const contexts</h4> + +<p> + In the default field values RFC, it was decided that default values for + structs must be const. The reasoning is that functions with side-effects may + be confusing for the compiler and for users. However, these justifications + don't apply to functions, because non-const functions don't carry an + expectation of being cheap or consistent. It is possible to always require + default values to be const, but there's seemingly no strong reason to do so. +</p> + +<p> + However, it would be confusing if default parameters could be non-const in + const functions, which is why they are required to be const in that case. +</p> + +<h4>Named arguments aren't mandatory</h4> + +<p>See the section above in "Drawbacks".</p> + +<h4>Alternative: Do nothing</h4> + +<p> + The "Motivation" section explains thoroughly how the current state is very + verbose, to the point where existing alternatives, such a struct parameters + and builders are often not used. The existence of these patterns is evidence + that users want some way to emulate named arguments. The fact that these + alternatives are not used in private APIs is evidence that the status quo is + annoyingly verbose. +</p> + +<h4>Alternative: Structural Records</h4> + +<p> + It has been proposed many times that structural records, alongside default + field values, could emulate named parameters. This is a pattern that is often + used in languages like TypeScript. The syntax for declaring such a function + would look like this. +</p> + +<pre> +fn foo({ a, b, c }: { a: f64, b: f64, c: f64 }) {} +</pre> + +<p>There are several criticisms I have of this pattern:</p> +<ul> + <li> + It cannot be backported to older functions without causing a breaking change, + so the standard library could not use it. + </li> + <li> + Every identifier in this declaration needs to be typed twice. It might be + possible to introduce syntactic sugar for this, but not such syntax has been + proposed. + </li> + <li> + Structural records have had a proposed RFC in the past, but was rejected, + mainly due to the complexity of implementing it. + </li> +</ul> + +<p> + Structural records could serve as a temporary compromise for named arguments, + if they are ever implemented. I'm not currently convinced that they ever will + be implemented. If the compromise solution is harder to implement than the + real solution, then we might as well just use the real solution. +</p> + +<h4>Alternative: Struct name inference</h4> + +<p> + One proposal to make named arguments easier is to allow the name of a struct + to be inferred when it is constructed. This would make the call sites of + functions using struct parameters to look mostly simple. +</p> + +<pre> +foo(_ { a: 1, b: "2", c: [] }); +</pre> + +<p> + However, this still requires the function to support this function explicitly, + by defining a struct to be passed in. It also still cannot be backported to + existing functions. The only benefit this provides over the status quo is to + avoid typing the name of the struct at the call site, which is not the main + criticism of this pattern. It would still be unlikely for private functions to + use this pattern. +</p> + +<h4>Alternative: Implementing <code>Fn</code> traits</h4> + +<p> + I've seen this come up a couple of times, but I think most of the people + suggesting this are doing so as a joke. Implementing <code>Fn</code> would + be a more verbose form of the method overload feature that some other + languages have. Overloads are difficult and controversial to implement in + Rust, so it's unlikely this pattern would be adopted. +</p> + +<h3>Prior art</h3> + +<p> + Several previous attempts at named arguments, both in Rust and in other + languages, have been described throughout this RFC. +</p> + +<p> + A previous RFC for named arguments exists for Rust. It was rejected mainly + because it did not describe what the future of named arguments would look like + with optional arguments. This RFC attempts to address as many future uses of + named arguments as possible in order to give a full picture. +</p> +<p> + A pre-RFC was proposed by forum user Azerupi, but no RFC seemed to come as a + result of it. Several complications were pointed out by commenters. This RFC + attempts to address syntactic ambiguity, the type system, and opt-in naming. +</p> +<p> + Another pre-RFC proposed using a more Swift-style syntax, which was avoided in + this RFC. The pre-RFC also proposed overloading but not default arguments. + Arguably the latter is more important than the former. +</p> +<p> + As was mentioned before, bon offers a procedural macro for creating builders + from functions. The existence of this feature shows that there is a desire for + named arguments, that would be better handled as a language feature. +</p> + +<p> + Other lanaguages with named arguments have also been discussed. The feature + proposed here is similar to named arguments from C# or Kotlin, with the + difference being that thhe named arguments are opt-in. The syntaxes used by + Dart and Swift have also been discussed. Dart uses the following syntax, which + looks closer to destructuring than named arguments. +</p> + +<pre> +void enableFlags({bool? bold, bool? hidden}) +enableFlags(bold: true) +</pre> + +<p> + Swift makes all arguments named by default, allowing authors to opt-out by + using an underscore before the variable name. This is impossible to port to + Rust without making a breaking change to the language. Various proposals have + tried to use slightly different syntaxes, but many of them present syntactic + ambiguities. +</p> + +<h3>Unresolved questions</h3> + +<ul> + <li>Should we require named arguments to be given in position order?</li> + <li>Is <code>pub</code> the best keyword, or should a different keyword be used?</li> + <li>Should default values be required to always be const?</li> +</ul> + +<h3>Future possibilities</h3> + +<h4>Default parameter values for trait methods</h4> + +<p> + Trait methods were forbidden from having default values due to the question of + if the default value should be on the trait declaration or the trait + implementation. Having it be on the implementation would be consistent with + how associated types and constants work. But this would make it hard for such + a trait to be dyn-compatible. +</p> + +<p> + One solution would be to only allow default values to be used in non-dyn + contexts. +</p> + +<pre> +trait Foo { + fn bar(&self, baz: i32 = 0); +} + +impl Foo for i32 { + fn bar(&self, baz: i32 = i32::MAX) {} +} + +fn do_bar<T: Foo>(x: &T) { + x.bar(); // baz doesn't need to be specified, since we're using generics +} + +fn do_bar_2(x: &dyn T) { + x.bar(); // ERROR: baz must be specified when using a method on a dyn object +} +</pre> + +<h4>Extending traits</h4> + +<p> + Theoretically, there's nothing preventing the following, as long as a default + parameter value is provided: +</p> + +<pre> +impl<T> Default for Vec<T> { + fn default(capacity: usize = 0) -> Self { + Self::with_capacity(capacity) + } +} +</pre> + +<p> + Although, in addition to making the behavior of the <code>default</code> + function inconsistent, this is even harder to resolve in a <code>dyn</code> + context. The likely remedy would be to forbid <code>Vec</code> from being + converted into a <code>dyn Default</code>, which would make introducing the + named parameter a breaking change. +</p> + +<h4>Requiring named arguments in a future edition</h4> + +<p> + Allowing named arguments to be optional is necessary in order to backport them + into existing functions without creating breaking changes. This comes with + some compromises, as has been described in the "Drawbacks" section. However, + the way this RFC is drafted makes it possible to require named arguments in a + future edition of Rust. Specifying a named argument would not give a warning + in either the previous edition or the new edition. +</p> + +<p> + Doing this too early would likely hurt adoption of named arguments, as + backporting them into existing functions would become a breaking API change. + This would also increase the verbosity of calling functions which make use of + named parameters, hurting adoption further. +</p> + +<h4>More flexible re-ordering</h4> + +<p> + As mentioned before, re-ordering of named arguments is allowed by this RFC, + in a limited way. The limitations are kept in order to keep the rules easy to + explain and reason about. However, some languages allow arguments to be + re-ordered more flexibly, and this would be possible to do in the future. +</p> + +<p> + C#'s main rule is that out-of-order named arguments cannot be followed by + positional arguments. +</p> + +<pre> +print_order_details(order_num: 31, product_name: "Red Mug", seller_name: "Gift Shop"); +print_order_details(seller_name: "Gift Shop", product_name: "Red Mug", order_num: 31); +print_order_details("Gift Shop", 31, product_name: "Red Mug"); +print_order_details(seller_name: "Gift Shop", 31, product_name: "Red Mug"); + +// this would cause an error +print_order_details(product_name: "Red Mug", 31, "Gift Shop"); +</pre> + +<h4>Defaults Affect Inference</h4> + +<p> + In 2022, Gankra proposed using default parameter values as part of inference + for generic type parameters. We could imagine something like this. +</p> + +<pre> +fn default_hasher() -> RandomState { ... } + +impl <K, V, S: BuildHasher> HashMap<K, V, S> { + fn new(hasher: S = RandomState::default()) { ... } +} + +// uses S=RandomState +let map = HashMap::new(); +</pre> + +<p> + The downside of this particular approach, as Gankra pointed out, is that this + would only allow the argument to be omitted if the type of <code>S</code> is + <code>RandomState</code>. Although, that's already true with the current set + of <code>HashMap</code> factories, unless you use <code>default</code> which + doesn't have any parameters at all. +</p> + +<h4>Generic const</h4> + +<p> + In the case of <code>Vec</code>, adding an optional <code>capacity</code> + parameter is not possible with this iteration of the RFC, because + <code>Vec::new</code> is a <code>const</code> function, and allocating memory + cannot be done at compile-time. This is a very difficult problem to solve, + and it's possible that it never will. However, if keyword generics are added + to the language, it may be possible to do something like this: +</p> + +<pre> +const<C> trait Capacity: Copy { + const<C> fn allocate<T>(self) -> Option<Box<[MaybeUninit<T>]>>; + + const fn as_usize(self) -> usize; +} + +const impl Capacity for () { + const fn allocate<T>(self) -> Option<Box<[MabyUninit<T>]>> { + None + } + + const fn as_usize(self) -> usize { + 0 + } +} + +impl Capacity for usize { + fn allocate<T>(self) -> Option<Box<[MaybeUninit<T>]>> { + Some(Box::default()) + } + + const fn as_usize(self) -> usize { + self + } +} + +const<C> impl<T, Cap: const<C> Capacity> Vec<T> { + pub const<C> fn new(capacity: Cap = ()) -> Self { + Self { + len: 0, + capacity: capacity.as_usize(), + buffer: capacity.allocate(), + } + } +} +</pre> + +<p> + If no argument value is provided, then the function would be + <code>const</code>. But when a <code>usize</code> is provided, then it would + no longer be <code>const</code>. +</p> + +</article> +</body> diff --git a/src/blog/quality-quantity.kuht b/src/blog/quality-quantity.kuht new file mode 100644 index 0000000..e3ddf99 --- /dev/null +++ b/src/blog/quality-quantity.kuht @@ -0,0 +1,98 @@ +<import "base.kuht" as "base" /> + +<head> + <title>More Features Don't Make Software Better</title> + <meta name="description" content="Something I should get off my chest" /> +</head> + +<body> +<article> + +<h1>More Features Don't Make Software Better</h1> + +<p> + Forgive the clickbaity title. It's not entirely true, but it is accurate to + the spirit of what I'm describing. I have frustration when explaining this to + people, and I've struggled to articulate it well until now. +</p> + +<p> + If your software has 100 features, but all of them suck, then I have no reason + to use your product. But if your software has even one feature that feels very + well considered, thought-out, and cared for, I now have a reason to use your + product. +</p> + +<p> + A similar principle applies when applying for jobs. If you send out 100 job + applications that all suck, then every single one of them is going to be + rejected. You can't scam someone into giving you a job. I spent two weeks + refining my resume, and only sent it to two companies. Both companies also got + cover letters from me. I got an offer from the second job I applied + for<a id="af-1" href="#footnote-1"><sup>1</sup></a>. +</p> + +<p> + My code editor of choice is helix. It doesn't have a lot of features. In fact, + it doesn't even support extensions yet<a id="af-2" href="#footnote-2"><sup>2</sup></a>. + But I keep coming back to it because the mental model it asks me to adopt is + very simple compared to something like Vim, and much more efficient than VS + Code. I select text, and then perform an action. The other things I need out + of a code editor are just good enough to get by. There is a built-in file + explorer and ripgrep tool. I've tried LazyVim in the past, and even though it + had more features, I couldn't shake the feeling that none of them were + designed to work nicely with each other. So despite how much functionality + there was, I didn't feel good about using any of it. +</p> + +<p> + Sonic Frontiers, after releasing, had three updates to the game that were + released for free. There were a lot of interesting ideas in those updates. + They did try to address some of the issues players had with the game, such as + the lack of momentum when jumping. But there were other ideas that I found + questionable. Action chain challenges are cool, but would it really improve + how players perceive the game if it were there at launch? None of the updates + improved the art direction, or addressed the gratuitous pop-in, or made the + existing Cyberspace stages feel less like copies of older levels. The final + update came with a new campaign at the end of the game with multiple playable + characters. The lack of multiple playable characters is a long-standing + complaint in the Sonic community, but the gameplay for these characters was so + difficult to control (with the exception of Amy) that I would rather have not + played as them at all. +</p> + +<p> + I don't think it should be acceptable to excuse having lots of bugs because + you also have a lot of features. It would make sense that more features would + induce more bugs, but that's still worse for the end-user. I'd rather have + software that doesn't do much, but also doesn't annoy me with problems while + it does very little. Those annoyances will cause some users to find other + options, even if those other options may not do everything your software does. +</p> + +<p> + All of this is my opinion. I won't claim that everyone else feels the same + way, or that following this approach will make you more money in the long run. + But personally, I prefer a piece of software that has one feature I really + like, to a piece of software that has 100 features that I only tolerate. +</p> + +</article> +<hr /> + +<footer> + <ol> + <li id="footnote-1"> + I decided not to do the interview for the first job, because it asked me to + record an interview where all of the questions would be asked by AI. I also + already had been preparing for interviews with the second company, so I + decided to focus on that. So, in effect, I only really applied for one job. + <a class="return" href="#af-1">return</a> + </li> + <li id="footnote-2"> + Helix is supposed to get plugin support soon, but it's been a long time. + <a class="return" href="#af-2">return</a> + </li> + </ol> +</footer> +</body> diff --git a/src/blog/try-catch-in-rust.kuht b/src/blog/try-catch-in-rust.kuht new file mode 100755 index 0000000..63734cd --- /dev/null +++ b/src/blog/try-catch-in-rust.kuht @@ -0,0 +1,553 @@ +<import "base.kuht" as "base" /> + +<head> + <title>How to Introduce try-catch to Rust</title> + <meta name="description" content="Don't do this, but the door is open for it." /> +</head> + +<body> + +<article> + +<h1>How to Introduce try-catch to Rust</h1> + +<p> + As a disclaimer, this isn't my personal proposal or anything. But I took a look at the RFC for try + blocks in Rust, and I thought it was interesting how it it left the door open for catch blocks in the + future. I thought it would be an interesting idea to share. But in order to talk about this, we have + to work our way forwards. +</p> + +<h2>Above the Iceberg: Rust enums</h2> + +<p> + Many of you readers will already understand this much, but if you've never worked with Rust before, + I'll have to introduce the enum system. Rust's enums can carry values on particular variants. They're + pretty much just tagged or discriminated unions. +</p> + +<pre> +pub enum JsonValue { + Null, + Boolean(bool), + Number(f64), + String(String), + Array(Vec<JsonValue>), + Object(Vec<(String, JsonValue)>), +} +</pre> + +<p> + There are a few ways we can get the inner value. We can use a match expression, which is a bit like + switch in other languages. +</p> + +<pre> +fn to_json(value: JsonValue) -> String { + match value { + JsonValue::Null => "null".to_string(), + JsonValue::Boolean(b) => b.to_string(), + JsonValue::Number(n) => n.to_string(), + JsonValue::String(s) => format!("\"{s}\""), + JsonValue::Array(a) => { + let mut builder = String::from("[ "); + for element in a { + builder.push_str(&to_json(element)); + builder.push_str(", "); + } + builder.push(']'); + builder + } + JsonValue::Object(o) => { + let mut builder = String::from("{ "); + for (key, value) in o { + builder.push_str(&format!("\"{key}\": {}", to_json(value))); + builder.push_str(", "); + } + builder.push('}'); + builder + } + } +} +</pre> + +<p>We can use if let to match on only one particular variant.</p> + +<pre> +fn unwrap_number(value: JsonValue) -> f64 { + if let JsonValue::Number(n) = value { + n + } else { + panic!("The passed value was not a number"); + } +} +</pre> + +<p> + More recently, we got let else, which lets us assert that it is one particular variant, or quit early. +</p> + +<pre> +fn unwrap_number(value: JsonValue) -> f64 { + let JsonValue::Number(number) = value else { + panic!("The passed value was not a number"); + } + + number +} +</pre> + +<p>But importantly, there's no way to get the inner value without first checking for the variant.</p> + +<h2>In the Beginning: The <code>Result</code> type</h2> + +<p> + Rust doesn't have exceptions. The <code>panic</code> macro, shown earlier, causes an immediate end to + the program (in most cases). For catchable errors, we use the <code>Result</code> type, which is + included in Rust's core library. +</p> + +<pre> +pub enum Result<T, E> { + Ok(T), + Err(E), +} +</pre> + +<p> + The <code>Result</code> type has two generic types: <code>T</code>, which is the type that was + expected to be returned; and <code>E</code>, which is the type that will be returned instead if an + error occurred. For example, we could define a <code>parse_number</code> function like this: +</p> + +<pre> +fn parse_to_f64(s: &str) -> Result<f64, ParseFloatError> +</pre> + +<p> + If the passed string is a number, then this function will return a 64-bit float. Otherwise, it will + return a <code>ParseFloatError</code>, which is an error describing a failure to parse the number. + We can use the result of this function like so: +</p> + +<pre> +fn foo() { + let number = match parse_to_f64("3.14159") { + Ok(number) => number, + Err(e) => panic!("{e}"), + }; + + println!("{number}"); +} +</pre> + +<p> + The <code>Result</code> type has a convenient helper method to do this, called <code>unwrap</code>. + If the result is <code>Ok</code>, then the inner value will be returned. Otherwise, the program will + panic. +</p> + +<pre> +fn foo() { + let number = parse_to_f64("3.14159").unwrap(); + println!("{number}"); +} +</pre> + +<p>But typically, you'd want to just throw the error again.</p> + +<pre> +fn foo() -> Result<(), ParseFloatError> { + let number = match parse_to_f64("3.14159") { + Ok(number) => number, + Err(e) => return Err(e), + }; + + println!("{number}"); +} +</pre> + +<h2>Introduced Later: The <code>?</code> operator</h2> + +<p> + As you might imagine, using that match expression over and over again to propogate errors will + quickly become tedious. The original solution was to use the <code>try</code> macro to do it for you. +</p> + +<pre> +fn foo() -> Result<(), ParseFloatError> { + let number = try!(parse_to_f64("3.14159")); + println!("{number}"); +} +</pre> + +<p> + But the 2018 edition of Rust introduced some syntactic sugar to make this even easier. It can be used + to propogate an error in any function which returns a similar <code>Result</code>. +</p> + +<pre> +fn foo() -> Result<(), ParseFloatError> { + let number = parse_to_f64("3.14159")?; + println!("{number}"); +} +</pre> + +<p>It can even convert between errors, if possible.</p> + +<pre> +enum JsonError { + ParseNumber(ParseFloatError), + Other(String), +} + +impl From<ParseFloatError> for JsonError { + fn from(value: ParseFloatError) -> Self { + Self::ParseNumber(value) + } +} + +fn foo() -> Result<(), JsonError> { + // automatically converts the ParseFloatError to a JsonError + let number = parse_to_f64("3.14159")?; + println!("{number}"); +} +</pre> + +<p>It also works on the <code>Option</code> type, which is also included in Rust.</p> + +<pre> +fn foo() -> Option<i32> { + let array = [1, 2, 3, 4]; + let x = array.get(4)?; // get the element at index 4, which doesn't exist in the array + Some(x) +} +</pre> + +<p> + And some of you may be surprised to learn that it also works on <code>ControlFlow</code> and some + <code>Poll</code> types. +</p> + +<p> + This isn't stable yet, but there's also a proposal to allow the use of the <code>?</code> operator on + custom types. This would be done using a few new traits: <code>Residual</code>, + <code>FromResidual</code>, and <code>Try</code>. A residual is a type that can be returned from using + a <code>?</code>. The <code>FromResidual</code> trait is implemented on types that can be created + from residuals (for example, using <code>?</code> to get a + <code>Result<Infallible, JsonError></code>, then converting that to a + <code>Result<(), JsonError></code>). The <code>Try</code> trait is implemented when a + <code>FromResidual</code> type can also have the <code>?</code> operator applied to it. These traits + are already available in nightly Rust. As an exercise for the Dallas Rust User Group, we created + <a href="https://github.com/joshuaPurushothaman/diy-result">our own <code>Result</code> type</a>, and + used these traits to make the <code>?</code> operator work with it. +</p> + +<h2>In Nightly: Try Blocks</h2> + +<p> + Today, <code>try</code> is a reserved keyword. So to use the <code>try!</code> macro in newer + editions of Rust, you'll need to use the raw-literal syntax. +</p> + +<pre> +fn foo() -> Result<(), ParseFloatError> { + let number = r#try!(parse_to_f64("3.14159")); + println!("{number}"); +} +</pre> + +<p> + What's the keyword being used for? It's for the try-block feature, which is currently only available + in nightly. The idea for this was made at roughly the same time as the <code>?</code> operator, to + catch anything that was propogated with it. However, only the <code>?</code> got merged. +</p> + +<pre> +let x: Result<_, _> = try { foo()?.bar()?.baz()? }; +</pre> + +<p>This way, you can catch an error without having to create a whole new function.</p> + +<h2>The Future: Catch blocks</h2> + +<p> + From what I understand, the original idea was to not reserve <code>try</code> as a keyword. Instead, + the keyword was going to be <code>catch</code>. That idea was dropped, presumably because it would + break with the convention created by every other programming language. But this also leaves the door + open to introduce catch blocks later. +</p> + +<p> + I haven't found any proposed RFCs for what catch blocks might look like. I think most people don't + want to think about this until <code>try</code> is finished. But people have noted that we have the + option of adding catch later, and the rough syntax isn't too hard to figure out. +</p> + +<pre> +let x: Result<f64, _> = try { + //... +} catch error { + panic!("Unexpected error: {:?}", error); +} +</pre> + +<p>We could pattern match on the expression too.</p> + +<pre> +let x: f64 = try { + //... +} catch JsonError::ParseNumber(error) { + panic!("Failed to parse float: {:?}", error); +} catch JsonError::Other(error) { + panic!("An unexpected error occurred: {:?}", error); +} +</pre> + +<p>We'd also have to allow try-catch on <code>Option</code>.</p> + +<pre> +let array = [1, 2, 3, 4]; +let x = try { + array.get(4)?; +} catch { // note that there's no reason for a variable to be created here + // this would probably desugar to `catch _` + // that desugaring would allow this syntax to also be used with Result + panic!("wait, what am i doing with my life?"); +} +</pre> + +<p> + Creating this API would probably also mean we'd have to think about adding a <code>Catch</code> + trait, so that it could be allowed on custom types. Modeling this is difficult, but can be done. +</p> + +<pre> +pub trait Catch: Try { + type Error; + + fn error(residual: Self::Residual) -> Self::Error; +} + +impl<T, E> for Result<T, E> { + type Error = E; + + fn error(residual: Self::Residual) -> Self::Error { + let Err(error) = self; + error + } +} + +impl<T> for Option<T> { + type Error = (); + + fn error(residual: Self::Residual) -> Self::Error { + () + } +} +</pre> + +<p>In case you're wondering about the desugaring:</p> + +<pre> +let x: f64 = match try { /* ... */ }.branch() { + ControlFlow::Continue(output) => output, + ControlFlow::Break(residual) => match Result::error(residual) { + JsonError::ParseNumber(error) => panic!("Failed to parse float: {:?}", error), + JsonError::Other(error) => panic!("An unexpected error occurred: {:?}", error), + } +} +</pre> + +<p> + Does that mean that <code>finally</code> could be added in the future too? Perhaps. I've never been a + big fan of finally blocks. But we can do it. If you're familiar with the +<a href="https://lib.rs/scopeguard">scopeguard</a> crate, then it shouldn't seem too difficult. +</p> + +<pre> +let x: f64 = try { + //... +} catch JsonError::ParseNumber(error) { + panic!("Failed to parse float: {:?}", error); +} catch JsonError::Other(error) { + panic!("An unexpected error occurred: {:?}", error); +} finally { + println!("finally! we're done!"); +} + +// desugars to... + +let x: f64 = { + defer! { + println!("finally! we're done!"); + } + + match try { /* ... */ }.branch() { + ControlFlow::Continue(output) => output, + ControlFlow::Break(residual) => match Result::error(residual) { + JsonError::ParseNumber(error) => panic!("Failed to parse float: {:?}", error), + JsonError::Other(error) => panic!("An unexpected error occurred: {:?}", error), + } + } +} +</pre> + +<h2>Yeet! We can go even further</h2> + +<p> + Now that we have most of the syntax from languages that have exceptions, let's go even further. + We could consider adding the <code>throw e</code> keyword as syntactic sugar for + <code>return Err(e)</code>. +</p> + +<pre> +let x: f64 = try { + throw JsonError::Other("oh no, an error!".to_string()); +} catch JsonError::ParseNumber(error) { + panic!("Failed to parse float: {:?}", error); +} catch JsonError::Other(error) { + panic!("An unexpected error occurred: {:?}", error); +} finally { + println!("finally! we're done!"); +} +</pre> + +<p> + This isn't available in nightly, because <code>throw</code> isn't yet a keyword in Rust. Worse, we + can't add it as a keyword until everybody agrees on what keyword to use. Some languages, such + Python, use <code>raise</code> instead. +</p> + +<p> + This problem is known as bikeshedding. People can't agree on what color to paint the bikeshed, so the + bikeshed never gets built. Everybody agrees that having the bikeshed is important, regardless of what + color it is. A common solution to this is to pick an option nobody would ever want. This lets everyone + know that we're not actually going to use that color, so we can and will change it later. This + happened. +</p> + +<pre> +#![feature(yeet_expr)] + +fn throw_error() -> Result<Infallible, JsonError> { + do yeet JsonError::Other("an error has been yeeted"); +} +</pre> + +<p> + Yep, <code>yeet</code> is the chosen keyword. The <code>do</code> needs to be there because + <code>yeet</code> is not a reserved word, but <code>do</code> is, despite it not being used for + anything. +</p> + +<p> + The way this works under the hood, is <code>do yeet e</code> desugars to <code>Yeet(e)</code>. + Then a couple of <code>FromResidual</code> implementations allow the <code>Yeet</code> type to be + used for error propogation. +</p> + +<pre> +impl<T> FromResidual<Yeet<()>> for Option<T> { + fn from_residual(_: Yeet<()>) -> Option<T> { + None + } +} + +impl<T, E, F: From<E>> FromResidual<Yeet<E>> for Result<T, F> { + fn from_residual(error: Yeet<E>) -> Result<T, F> { + Err(error.0.into()) + } +} +</pre> + +<p> + Now that we've been using <code>yeet</code> for so long, there are some people who want this to be + the new keyword. There was even a proposal to reserve yeet for Rust 2021. It was rejected. +</p> + +<h2>Now in Reverse? What?</h2> + +<p> + I had to share this because Rust does not have exceptions. The idea that we could introduce an + exception-like syntax for a language which doesn't have them is amazing to me. +</p> + +<p>This got me thinking, could we introduce <code>Result</code> to a language that doesn't have it?</p> + +<p>Imagine the following C# code, where using try, but without a catch, returns a <code>Result</code>.</p> + +<pre> +// I used a wildcard for the error type because C# can throw one of multiple exceptions here +Result<decimal, _> = try { + yield demical.Parse("3.14159"); +} +</pre> + +<p> + This would be easier in languages where block expressions exist. If a language were to pursue this, + I'd prefer they call the type <code>Except</code> or <code>Exceptional</code> instead of + <code>Result</code>. +</p> + +<h2>Conclusion</h2> + +<p> + The main I reason I wrote this post was because, when I proposed a new scripting language in an + <a href="/blog/best-tool-for-the-job.html">earlier blog post</a>, I wanted to talk about how + try/catch would be implemented in it. But I figured that was way too off-topic for that post. So I + wanted to discuss it here. +</p> + +<p> + My scripting language would have unchecked exceptions, with the option to catch errors and put them + inside of an <code>Exceptional</code> type. +</p> + +<pre> +function throws() { + throw "oh no" +} + +throws() // this crashes the script +</pre> + +<p>Wrapping an expression in a try block yields an <code>Except</code> type.</p> + +<pre> +function throws() { + fail with "oh no" +} + +let x: Except<any, any> = try { + throws() // this doesn't crash +} +</pre> + +<p>I think it's feasible to use a postfix <code>!</code> as syntactic sugar for this</p> + +<pre> +function throws() { + fail with "oh no" +} + +let x: Except<any, any> = throws()! +</pre> + +<p>Then with <code>catch</code> or <code>.catch</code>, the error can be handled</p> + +<pre> +function throws() { + fail with "oh no" +} + +let x: Except<any, any> = throws()!.catch error { + print("an error occurred") + exit() +} +</pre> + +<p>I like this idea. I'm glad I could share it with you.</p> + +</article> +</body> diff --git a/src/blog/unwind-safety-happylock.kuht b/src/blog/unwind-safety-happylock.kuht new file mode 100755 index 0000000..61c36c8 --- /dev/null +++ b/src/blog/unwind-safety-happylock.kuht @@ -0,0 +1,555 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Poisoning, Unwind Safety, Exception Safety, and More in HappyLock</title> + <meta name="description" content="HappyLock 0.4 is now out! This is a journey in all things related to safety and panics." /> +</head> + +<body> + +<article> + +<h1>Poisoning, Unwind Safety, Exception Safety, and More in HappyLock</h1> + +<p> + I'm proud to announce that HappyLock 0.4 has been released. I thought that I should focus on + poisoning because I thought, at the time, that it would be simple. I was very wrong. Poisoning + exposed a whack-a-mole of challenges to solve, which I will illustrate here. +</p> + +<h2>A Refresher on HappyLock</h2> + +<p> + I recommend reading the full article on HappyLock. I've been told it's a very good read. But I'll + briefly explain the idea here. Those of you who have already read the full post can skip ahead to + the new stuff. +</p> + +<p> + HappyLock uses the Rust borrow checker to make deadlocks impossible at compile-time. This is done by + having a <code>ThreadKey</code> which is not <code>Send</code>, <code>Copy</code>, or + <code>Clone</code>. But it is now <code>Sync</code>. Only one <code>ThreadKey</code> can exist + on each thread at a time. In order to lock anything, you must pass either a <code>ThreadKey</code> + or a <code>&mut ThreadKey</code> into the <code>lock</code> function. This means that in order to + lock anything new, you must first release any resources you have already acquired, preventing partial + allocation. +</p> + +<pre> +fn main() { + // This panics if the ThreadKey has already been acquired for this thread + let mut key = ThreadKey::get().unwrap(); + + let mutex1 = Mutex::new(42); + let mutex2 = Mutex::new(76); + + thread::spawn(|| { + // You're allowed to have more than one ThreadKey, but not on the same thread + let key = ThreadKey::get().unwrap(); + + let guard = mutex.lock(&mut key); // Note the key. Nothing else can use it now + *guard = 24; + drop(guard); // Now we can use the ThreadKey again + + let guard = mutex2.lock(&mut key); + *guard = 67; + }).join().unwrap(); + + let guard = mutex.lock(&mut key); + assert_eq!(guard, 24); + // the guard is implicitly dropped here + let guard = mutex.lock(&mut key); + assert_eq!(guard, 67); +} +</pre> + +<p> + If you want to lock multiple mutexes at the same time, you can use a <code>LockCollection</code>, + which will ensure that all locks are acquired at the same time, so there's no possibility of deadlock. + By default, this works by sorting all of the locks by their memory address, preventing circular wait, + but there are other ways to do this with different performance qualities. +</p> + +<pre> +fn main() { + let key = ThreadKey::get().unwrap(); + let collection = LockCollection::new((Mutex::new(42), Mutex::new("foo"))); + let guard = collection.lock(key); + assert_eq!(guard.0, 42); + assert_eq!(guard.1, "foo"); +} +</pre> + +<p> + Lots of things can be used in a <code>LockCollection</code>, which have the <code>Lockable</code> trait. + There's also the <code>RawLock</code> trait, which is implemented for <code>Mutex</code> and + <code>RwLock</code>. +</p> + + +<h2>What is Unwind Safety</h2> + +<p> + Unwind safety has nothing to do with the conventional notion of safety in Rust. And although + object-safety has recently been rebranded to dyn-compatible, unwind safety cannot be. The notion + of unwind safety is actually the entire reasoning why poisoning exists. Imagine the following + incredibly contrived scenario. +</p> + +<pre> +struct MultipleOfFive(i8); + +impl MultipleOfFive { + fn increment(&mut self) { + catch_unwind(|| { + for _ in 0..5 { + self.0 += 1; + } + }); + } +} +</pre> + +<p> + The variant that is supposed to be held here is that a <code>MultipleOfFive</code> is always + divisible by five. But what happens if debug assertions are enabled, and our value is 125? +</p> + +<pre> +125 + 1 = 126 +126 + 1 = 127 +127 + 1 ??? integer overflow! panic! +</pre> + +<p> + But, we caught that panic, so now the value will always be 127, even though 127 is not divisible + by five. This doesn't result in undefined behavior. It's just very bizarre. This is the purpose + of the <code>UnwindSafe</code> trait. It asserts that unwinding won't result in unexpected behavior + for the type. In fact, the code from before doesn't compile. +</p> + +<pre> +struct MultipleOfFive(i8); + +impl MultipleOfFive { + fn increment(&mut self) { + catch_unwind(|| { + for _ in 0..5 { + self.0 += 1; // compile error: mutable references are not unwind-safe + } + }); + } +} +</pre> + +<p> + But again, this cannot be used to prevent undefined behavior. To make this point clear, there + is a wrappper type in the standard library that actually lets us work around unwind-safety. We + can just use <code>AssertUnwindSafe</code>, which is completely safe to construct. +</p> + +<pre> +struct MultipleOfFive(i8); + +impl MultipleOfFive { + fn increment(&mut self) { + let this = AssertUnwindSafe(self); + catch_unwind(move || { + for _ in 0..5 { + (*this).0 += 1; + } + }); + } +} +</pre> + +<p> + So that begs the question: if mutable references are not unwind safe, why is <code>Mutex</code> + unwind safe? That's because of poisoning. Any time when a panic is caught while a + <code>MutexGuard</code> is in scope, the mutex gets poisoned. Any future calls to <code>lock</code> + on that mutex will return a <code>PoisonError</code>. Then, it becomes the responsibility of the + caller to decide how to handle that error. Most people don't care and just `unwrap` the result. + Most people seem to think that providing poisoning by default in the standard library was a mistake. + Several locking crates, like parking lot and spin, don't include poisoning at all. There's even some + work being done to provide non-poisoning variations of the locking primitive in the standard library. + For HappyLock, I want to provide as many of the standard library's features as possible, so I worked + out an alternative approach where you can get poisoning, but not by default. +</p> + +<h2>How to Poison a Mutex</h2> + +<p> + I decided to implement poisoning in the simplest way I could think of. HappyLock is already + filled with wrappers, so why not use one here too? Put simply, the implementation looks like + this: +</p> + +<pre> +struct Poisonable<L: RawLock + Lockable> { /* ... */ } + +impl Poisonable<L> { + // this is way over-simplified but you probably get the point + pub fn lock(&self) -> PoisonGuard<L::Guard>; +} +</pre> + +<p> + Unlike with <code>LockCollection</code>, <code>Poisonable</code> requires that <code>L</code> be + a <code>RawLock</code>. That's because <code>Poisonable</code> can't decide on its own how to deal + with multiple locks without deadlocking. That's the purpose of <code>LockCollection</code>. But, + <code>RawLock</code> already has a <code>lock</code> method, so it's easy to just call that to + lock whatever we need. +</p> + +<p> + At the same time, it would be nice if we could put a <code>LockCollection</code> inside of a + <code>Poisonable</code>. So, <code>LockCollection</code> now implements <code>RawLock</code>, + so you can create a <code>Poisonable<LockCollection<Vec<Mutex<String>>>></code> + and not unwrap every single mutex. +</p> + +<p> + There's another method that the standard library's <code>Mutex</code> has that is much harder to + implement: <code>get_mut</code>. This is already implemented on HappyLock's <code>Mutex</code>. It + also exists on the lock collections, but instead of getting mutable references to the data being + locked, it actually returns the collection. For example, + <code>LockCollection<Vec<Mutex<i32>>>::get_mut</code> returns a + <code>&mut Vec<Mutex<i32>></code>. Doing the same thing on a <code>Poisonable</code> would + require calling <code>get_mut</code> twice where the standard library only requires it to be called + once. +</p> + +<p> + So, I've made a couple more traits: <code>LockableGetMut</code> and <code>LockableIntoInner</code> +</p> + +<pre> +trait LockableGetMut { + type Inner<'a>; + + fn get_mut(&mut self) -> Self::Inner<'_>; +} + +trait LockableIntoInner { + type Inner; + + fn into_inner(self) -> Self::Inner; +} +</pre> + +<p> + You might wonder, why not just have the same <code>Inner</code> type on both traits, and then we can + have <code>get_mut</code> return <code>&mut Self::Inner</code>. But consider the case of a tuple. + In order to have a method like <code>get_mut</code>, we have to create a new tuple with a different + type that has no mutexes. But we can't return a mutable reference to a tuple we just created! But, + we <em>can</em> implement the function quite simply with this system. +</p> + +<pre> +impl<A: LockableGetMut, B: LockableGetMut> LockableGetMut for (A, B) { + type Inner<'a> = (A::Inner<'a>, B::Inner<'a>); + + fn get_mut(&mut self) -> Self::Inner<'_> { + (self.0.get_mut(), self.1.get_mut()) + } +} +</pre> + +<p> + For consistency, lock collections now also use this functionality, and the previous <code>into_inner</code> + method is now called <code>into_child</code>. This is a breaking change, but there are other breaking + changes that we'll also need to make here, as we'll soon see. +</p> + +<h2>The Exception Safety of HappyLock</h2> + +<p> + Exception safety is a less optional form of unwind safety. The Rustonomicon actually refers to + unwind-safety as "maximal exception safety". But even if we don't always guarantee that safe code + will have the correct behavior, we need to at least make sure that our types are exception-safe + to the point where unwinding does not cause potential unwind safety. Here's an example of how this + could be violated. +</p> + +<pre> +impl<T: Clone> Vec<T> { + fn push(&mut self, x: &T) { + self.reserve(1); + unsafe { + self.set_len(self.len() + 1); + self.ptr().add(1).write(x.clone()); + } + } +} +</pre> + +<p> + The problem is that <code>clone</code> can panic, because you can implement it however you want. + If it panics, then the length will be incorrect forever. This is something that you'll always need + to be careful about in unsafe code. Now, look at this function in HappyLock. +</p> + +<pre> +impl<L: OwnedLockable> OwnedLockCollection<L> { + pub fn lock<'g, 'key, Key: Keyable + 'key>( + &'g self, + key: Key, + ) -> LockGuard<'key, L::Guard<'g>, Key> { + let locks = get_locks(&self.data); + for lock in locks { + // safety: we have the thread key, and these locks happen in a + // predetermined order + unsafe { lock.lock() }; + } + + let guard = unsafe { self.data.guard() }; + LockGuard { + guard, + key, + _phantom: PhantomData, + } + } +} +</pre> + +<p> + As we can see, it is considered undefined behavior in HappyLock to cause a deadlock. Now, what + would happen if we had a collection of three locks, and the second one panicked. The first mutex + would never unlock, and then we could call it again, causing a deadlock, like this. +</p> + +<pre> +fn main() { + let collection = OwnedLockCollection::new(( + Mutex::new(1), + Mutex::new(2), + Mutex::new(3) + )); + + catch_unwind(|| { + let key = ThreadKey::get().unwrap(); + collection.lock(); + // We'll say for the sake of argument that our second mutex panics here + // The first mutex was never unlocked + }); + + // The ThreadKey was still dropped, so we can re-acquire it. + let key = ThreadKey::get().unwrap(); + + // But wait, the first mutex is still locked by this thread. + collection.lock(); // DEADLOCK!!! +} +</pre> + +<p> + Note that this only becomes a problem if, somehow, parking_lot panics when calling the + <code>lock</code> function. This would be considered a bug, but since you can use whatever + <code>RawMutex</code> you want with HappyLock, I can't guarantee that this will never happen. + This could be solved by just requiring a specific underlying API to be used, but I'm not ready + to impose that restriction (yet). Somebody at RustConf suggested using a fancy guard insertion + strategy so that the mutexes will automatically unlock on panic. That might work, but I have + a simpler solution. +</p> + +<p> + We can use the <code>scopeguard</code> library to automatically run code when unwinding. What + we'll need to do is keep track of which locks have been locked so far, and then, if we start + unwinding, unlock them. +</p> + +<p> + Now, if you use the <code>scopeguard</code> crate, this will actually result in a stack overflow, + depending on how rigourously you check for this. It won't check to make sure that the panic was + caused by your function, so it will keep calling itself, and you'll get a stack overflow. So I + made my own utility to check that the unwinding was caused by a specific piece of code +</p> + +<pre> +fn handle_unwind<R, F: FnOnce -> R, G: FnOnce()>(try_fn: F, catch: G) -> R { + let try_fn = AssertUnwindSafe(try_fn); + catch_unwind(try_fn).unwrap_or_else(|e| { + catch(); + resume_unwind(e) + }) +} +</pre> + +<p>Then we can use it to unlock on failure, like so</p> + +<pre> +unsafe fn ordered_lock(locks: &[&dyn RawLock]) { + let locked = Cell::new(0); + + handle_unwind( + || { + for lock in locks { + lock.raw_lock(); + locked.set(locked.get() + 1); + } + }, + || attempt_to_recover_locks_from_panic(&locked[0..locked.get()]), + ) +} + +unsafe fn attempt_to_recover_locks_from_panic(locked: &[&dyn RawLock]) { + handle_unwind( + || locked.for_each(|lock| lock.raw_unlock()), + || locked.for_each(|l| l.poison()), + ) +} +</pre> + +<p> + Ok, so I lied when I said there's no poisoning by default. It's not documented anywhere, but yeah. + If the unlock function panics, when we're kinda screwed, so all <code>RawLock</code>s now have to + be poisonable. This is a different kind of poisoning, which causes all future calls to the mutex to + panic, similar to the poisoning of <code>Once</code>. It's not exposed in the signatures for any of + the functions. I don't want you to ever have to think about this kind of poisoning. I emphasize, + this should never happen. If it does, then the <code>RawMutex</code> probably contains a bug that + already needs to be fixed. +</p> + +<p> + The lock that panicked is poisoned, regardless of whether it was locked or not. That's because I have + zero clue whether or not it actually locked. I would hope that if the <code>lock</code> function + panics, it didn't lock, but I don't know that for sure. I also don't have a good way to check, without + excluding support for certain mutexes in the future. Someone might suggest using the result of + <code>try_lock</code> to check if it's locked or not, and then unlock it in either case. That person + might be surprised to learn that, in C, it is undefined behavior to call <code>try_lock</code> on + a mutex that is already acquired by the calling thread. So there's not much I can do other than panic + here. +</p> + +<h2>Miscellaneous Changes</h2> + +<p> + The <code>try_lock</code> function now returns a <code>Result</code> instead of an <code>Option</code>. + This is something I've been thinking about for a while now, and the fact that <code>Poisonable</code> + now exists makes an even more compelling argument for using <code>Result</code>. If something + is already locked, then you can have the key back without having to call <code>ThreadKey::get</code>. +</p> + +<p> + I've kinda shown this already in some of the examples, but the <code>RawLock</code> trait methods have + been renamed. For example, <code>RawLock::lock</code> is now <code>RawLock::raw_lock</code>. +</p> + +<p> + Some of the reading APIs have been moved from <code>Lockable</code> to <code>Sharable</code>. The reason + they were in <code>Lockable</code> before was because I had an idea where you could mix sharables + with non-sharables in a lock collection, and then calling <code>read</code> would just get an exclusive + lock if there was no shared lock available for an entry. Eventually I decided this was a bad idea, but + I still had the <code>read_guard</code> method in <code>Lockable</code>. So now it's moved to + <code>Sharable</code>. +</p> + +<p> + I mentioned this already in the introduction to HappyLock, but <code>ThreadKey</code> is now + <code>Sync</code>. This is possible for the same reason that the nightly <code>Exclusive</code> + type is <code>Sync</code>. A <code>&ThreadKey</code> is completely useless, so no undefined + behavior can be caused because of it. This also allows guards which hold a key (i.e. <code>MutexGuard</code>) + to be <code>Sync</code>. They still can't be send because, among other things, dropping a + <code>ThreadKey</code> on the wrong thread can lead to having multiple <code>ThreadKey</code>s + on the same thread, making deadlock possible. The Rust standard library also doesn't implement + <code>Send</code> on its guard types, so I think this is acceptable. +</p> + +<p> + I'm now using cargo-diet before I publish HappyLock, so examples are no longer included when you + download it. +</p> + +<p> + The <code>ThreadKey</code> now uses a <code>Cell<bool></code> instead of an <code>AtomicBool</code> + when deciding if you can get a key or not. It's a thread-local variable, so there was no + synchronization needed in order to access the value. +</p> + +<p> + Speaking of which, I've removed the <code>thread-local</code> crate as a dependency and learned + enough to just use the standard library's <code>thread_local</code> macro. It's more efficient + anyway. Also, now that <code>LazyCell</code> is in the standard library, I've dropped the + <code>once_cell</code> dependency. But this does increase the MSRV substantially. +</p> + +<p> + The guard and ref types now implement more traits like <code>Eq</code>, <code>Ord</code> and + <code>Hash</code>. This would be a breaking change in the standard library, but this is already + a breaking change here. +</p> + +<p> + I've been using <code>cargo-mutants</code> to help me write better unit tests. This helped me + catch one bug before it was even released. Although, I did still find a bug, so it's not perfect. + HappyLock is in this weird state where it's being used and I'm scared of obvious bugs, but it's + not used enough to the point where obvious bugs are immediately found and fixed. I should + probably start using <code>tarpaulin</code> as well. +</p> + +<h2>The Future</h2> + +<p> + I'm still deciding what my next course of action will be for this crate. I should write more tests + for it before I do another big release. The documentation examples could definitely be improved. Some + of my plans for the future have changed. I still would like to try copying the standard library + lock implementations at some point. I no longer think compile-time duplicate checks are possible + due to the lack of const trait methods in Rust. LockCell and ReadOnlyLockCollection are still just + ideas in the back of my mind. But I have acquired a few new ideas since last year. +</p> + +<h3>Once</h3> + +<p> + It is definitely possible to make a deadlock-free <code>Once</code>, by having <code>call_once</code> + require a <code>ThreadKey</code>. I tried somehow fitting it into a <code>LockCollection</code>, but + I don't think that's practical, and I don't know why you would ever want more than a single + <code>Once</code> at a time. Although, I can see a point to using a <code>Mutex</code> inside of a + <code>call_once</code>, so I can also provide a way to pass in a <code>RawLock + Lockable</code> with it. +</p> + +<p> + I'm still not sure if I want to include <code>Condvar</code>. It is useful but I don't think I would + be able to prevent it from deadlocking. I'll keep it in mind though. +</p> + +<h3>Async</h3> + +<p> + During my time at RustConf, I realized that the biggest barrier to HappyLock adoption is probably + the inability to use it with async Rust. The problem is that you can have multiple tasks running on + the same thread, so <code>ThreadKey::get()</code> will return <code>None</code> very often. But a + good solution to this would be to have <code>TaskKey</code>s instead, which would only be useful + on the async mutexes. But this is something that the async runtime would need to be able to provide. + But if the <code>spawn</code> function gave you a <code>TaskKey</code>, then there would be no need + to even call a <code>get</code> function for it. The biggest problem here would be that you can't use + a blocking lock, because if it's already locked in a task that runs on the same thread, then this + would cause a deadlock. +</p> + +<h3>Expanding Cyclic Wait</h3> + +<p> + A reddit comment gave me an idea on how to pursue this. I don't think I'll go into too much detail + right now, but it would involve a <code>IntoLockIterator</code> trait, several new types, adding + a <code>&ThreadKey</code> to all of the ref guard types, and quite a bit of type-state analysis. + I'll probably talk about this over at the Miami Rust User Group, so join us online if you're + interested. I think we can turn HappyLock from a library that primarily prevents partial allocation + and prevents cyclic wait as an optimization, to a library that is used to prevent cyclic wait + and uses remnants of total allocation to prevent you from screwing up. +</p> + +</article> + +<hr/> + +<footer> + <ol> + <li id="footnote-1"> + I'm not quite sure what software engineering majors do at RIT. I imagine them all sitting + around a table, and saying words like "agile" and "scrum" with no context. I took Computer + Science. <a class="return" href="#af-1">return</a> + </li> + </ol> +</footer> + +</body> diff --git a/src/index.kuht b/src/index.kuht new file mode 100755 index 0000000..c0a3929 --- /dev/null +++ b/src/index.kuht @@ -0,0 +1,33 @@ +<import "base.kuht" as "base" /> + +<head> + <title>Botahamec's Website</title> +</head> + +<body> + <h1>Botahamec</h1> + + <p> + Hi. This is a website I threw together quickly so that I could write my blog post. It's not + much right now. But you should read my blog posts anyway. + </p> + + <h2>Projects</h2> + <ul> + <li><a href="/cgit/happylock">HappyLock</a> (Deadlock-free mutexes)</li> + <li><a href="/cgit/git-autosave">git-autosave</a> (Autosave for Git)</li> + <li><a href="/cgit/exun">Exun</a> (Expected and unexpected error handling)</li> + <li><a href="/cgit/snob">Snob</a> (Snobol-inspired string parsing)</li> + </ul> + + <h2>Blog</h2> + <ul> + <li><a href="/blog/how-happylock-works.html">How HappyLock Works</a></li> + <li><a href="/blog/best-tool-for-the-job.html">The Best Tool for the Job</a></li> + <li><a href="/blog/an-inheritance-detox.html">An Inheritance Detox</a></li> + <li><a href="/blog/try-catch-in-rust.html">Introducing try-catch to Rust</a></li> + <li><a href="/blog/comments-at-end.html">Comments at the End</a></li> + <li><a href="/blog/datetime.html">Comparing Date Types Across Languages</a></li> + <li><a href="/blog/color-spaces.html">Color Spaces</a></li> + </ul> +</body> diff --git a/style-dark.css b/style-dark.css new file mode 100755 index 0000000..fcb69de --- /dev/null +++ b/style-dark.css @@ -0,0 +1,22 @@ +body { + color: var(--white); + background-color: var(--black); +} + +a { + color: var(--light-blue); +} + +a.return { + color: var(--light-green); +} + +.nav { + background-color: var(--light-black); +} + +.nav__button { + color: var(--white); + background-color: var(--dark-purple); + border-color: var(--purple); +}
\ No newline at end of file diff --git a/style-light.css b/style-light.css new file mode 100755 index 0000000..a081c3c --- /dev/null +++ b/style-light.css @@ -0,0 +1,22 @@ +body { + color: var(--black); + background-color: var(--white); +} + +a { + color: var(--dark-blue); +} + +a.return { + color: var(--dark-green); +} + +.nav { + border-color: var(--dark-white); +} + +.nav__button { + color: var(--black); + background-color: var(--light-purple); + border-color: var(--purple); +} diff --git a/style.css b/style.css new file mode 100755 index 0000000..bfb0b2a --- /dev/null +++ b/style.css @@ -0,0 +1,102 @@ +body { + max-width: 800px; + margin: auto; + padding-left: 20px; + padding-right: 20px; + + --dark-red: #623a46; + --red: #8f6370; + --light-red: #bf8f9d; + + --dark-orange: #623e29; + --orange: #8f6752; + --light-orange: #be947d; + + --dark-yellow: #4f481f; + --yellow: #7a7349; + --light-yellow: #a7a074; + + --dark-green: #2f5136; + --green: #587c5f; + --light-green: #84a98b; + + --dark-sky: #145154; + --sky: #447c7f; + --light-sky: #71aaad; + + --dark-blue: #2e4a67; + --blue: #577594; + --light-blue: #82a2c3; + + --dark-purple: #4d4063; + --purple: #776a90; + --light-purple: #a496bf; + + --white: #efedf4; + --dark-white: #ceccd3; + --lighter-gray: #afadb3; + --light-gray: #908e94; + --gray: #727077; + --dark-gray: #56545a; + --darker-gray: #3b393f; + --light-black: #222126; + --black: #0c0a0f; +} + +pre { + margin-left: 25px; + overflow-x: auto; +} + +.nav { + margin: 18pt 0 18pt 0; + padding: 3px 15px; + border-width: 2px; +} + +.nav nav { + display: inline-block; +} + +.nav menu { + display: inline-block; + list-style-type: none; + padding-left: 0; +} + +.nav li { + display: inline-block; +} + +.nav__title { + display: inline-block; + font-size: 18pt; + margin-right: 18pt; +} + +.nav__button { + display: inline-block; + text-decoration: none; + margin-right: 3px; + border-width: 3px; + border-style: solid; + padding: 8px; +} + +.screen-reader-focusable { + position: absolute; + left: -10000px; + top: auto; + width: 1px; + height: 1px; + overflow: hidden; +} + +.screen-reader-focusable:focus { + position: unset; + left: unset; + top: unset; + width: unset; + height: unset; + overflow: unset; +} |
