commit - 182e3431233226e0209de750f784d79f6a4dedac
commit + 48182fda9a11903d8e5bc1df49e2a6f00deea33b
blob - 0d5452cdf5907dd381bd0e845f8e4e46646098fe
blob + 095f3684503a9f5661db7b6e347fca256f267d93
--- Cargo.lock
+++ Cargo.lock
[[package]]
name = "alpaca"
-version = "0.5.0"
+version = "0.6.0"
dependencies = [
"anyhow",
"assert_cmd",
[[package]]
name = "indexmap"
-version = "2.14.1"
+version = "2.14.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb"
+checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855"
dependencies = [
"equivalent",
"hashbrown",
[[package]]
name = "js-sys"
-version = "0.3.104"
+version = "0.3.105"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a"
+checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e"
dependencies = [
"cfg-if",
"futures-util",
[[package]]
name = "wasm-bindgen"
-version = "0.2.127"
+version = "0.2.128"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70"
+checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf"
dependencies = [
"cfg-if",
"once_cell",
[[package]]
name = "wasm-bindgen-futures"
-version = "0.4.77"
+version = "0.4.78"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "6b7777d5cc23d0e91404e53ce2d5e8ec7acae3026b16233dba62cd3246457950"
+checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928"
dependencies = [
"js-sys",
"wasm-bindgen",
[[package]]
name = "wasm-bindgen-macro"
-version = "0.2.127"
+version = "0.2.128"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1"
+checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed"
dependencies = [
"quote",
"wasm-bindgen-macro-support",
[[package]]
name = "wasm-bindgen-macro-support"
-version = "0.2.127"
+version = "0.2.128"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284"
+checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a"
dependencies = [
"bumpalo",
"proc-macro2",
"quote",
- "syn 2.0.119",
+ "syn 3.0.5",
"wasm-bindgen-shared",
]
[[package]]
name = "wasm-bindgen-shared"
-version = "0.2.127"
+version = "0.2.128"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf"
+checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e"
dependencies = [
"unicode-ident",
]
[[package]]
name = "web-sys"
-version = "0.3.104"
+version = "0.3.105"
source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "c435338968042f4f59a557f690a253676d47ce13ceb55d70100e7facf6620a30"
+checksum = "9fbddc4a036f00ec4f18c83445bd3115cb306a91da554919a099d9222fe4a7f8"
dependencies = [
"js-sys",
"wasm-bindgen",
blob - 5bf1d4db57792e827114536014f7b926f85d7824
blob + 2778703e010dc80903516e05ec2ed144be3103f0
--- Cargo.toml
+++ Cargo.toml
[package]
name = "alpaca"
authors = ["leoshimo", "mtmn"]
-version = "0.5.0"
+version = "0.6.0"
edition = "2024"
description = "Unix native interface for LLMs"
repository = "https://github.com/leoshimo/cogni"
blob - 11a471e935699a617c2d5715c19cb4b7c0fb58f8
blob + 8068673f35d24b281739faae2d7a9d74c764730b
--- README.md
+++ README.md
## alpaca quota
-`alpaca quota` asks a provider how much of your quota you have used. It prints
-each quota window as a bar showing the used share, with a countdown to the next
-reset. Where the provider reports per-model request counts, it prints those as a
+`alpaca quota` shows how much of your quota you have used. It asks the
+provider, then prints each quota window as a bar with a countdown to the next
+reset. If the provider reports per-model counts, it also prints those as a
table.
-Pick the provider with `-p` (default `ollama`):
+Pick the provider with `-p`. It defaults to `ollama`:
```sh
$ alpaca quota # Ollama Cloud
$ alpaca quota -p anthropic # Claude Code
$ alpaca quota -p openai # Codex CLI
+$ alpaca quota -p deepinfra # DeepInfra
```
Each provider has its own default credential and base URL:
| `ollama` | `$OLLAMA_API_KEY` | `$OLLAMA_API_ENDPOINT`, then `https://ollama.com` |
| `anthropic` | `$CLAUDE_CODE_OAUTH_TOKEN`, then `~/.claude/.credentials.json` | `https://api.anthropic.com` |
| `openai` | `$CODEX_ACCESS_TOKEN`, then `~/.codex/auth.json` | `https://chatgpt.com` |
+| `deepinfra` | `$DEEPINFRA_API_KEY` | `https://api.deepinfra.com` |
-The last two read the credential file the vendor CLI writes when you log in.
-They work as soon as you have logged into that CLI. Those endpoints serve the
-vendor CLIs and are not documented public APIs, so they may change without
-notice. Both expect a short lived OAuth token: if the credential file is stale,
+For Anthropic and OpenAI, alpaca reads the credential file the vendor CLI
+writes when you log in. Log in to that CLI once and it works. Those endpoints
+serve the vendor CLIs, not a documented public API, so they may change without
+notice. Both use a short-lived OAuth token. If the credential file goes stale,
you get an authentication error until the vendor CLI refreshes it.
+DeepInfra bills per token or per second, with no fixed quota. So alpaca prints
+no window for it. Instead, it prints the current month's spend as a note, and
+billed units per model as a table.
+
+The top-level `apikey` and `base_url` settings apply to Ollama only. To use
+them with Anthropic, OpenAI or DeepInfra, declare them in a profile and select
+it with `--profile`. A profile chosen automatically through `default_profile`
+does not apply to those providers. Command-line options and
+`API_KEY`/`API_ENDPOINT` override any profile.
+
+DeepInfra also reads a key from the top-level `deepinfra_apikey` setting, with
+no profile needed. An explicit profile's `apikey` still overrides it.
+
Options:
-- `-p, --provider <PROVIDER>`: one of `ollama`, `anthropic`, `openai`
+- `-p, --provider <PROVIDER>`: one of `ollama`, `anthropic`, `openai`, `deepinfra`
- `-T, --timeout <SECS>`: request timeout in seconds
- `--apikey <KEY>`: API key or OAuth token, replacing the provider default
- `--base-url <URL>`: base URL, replacing the provider default
blob - 371fda1c82d78e9dc50af7835131473d8c4c2f8a
blob + 968d4d7596eb6197fb4ba69442a9bd3728dafac5
--- config.example.toml
+++ config.example.toml
base_url = "https://ollama.com"
reasoning_effort = "none"
default_profile = "ollama-cloud"
+deepinfra_apikey = "your-deepinfra-key"
[ollama-cloud]
apikey = "your-ollama-cloud-key"
blob - 6f75d4b9e0d69798da77a91723c0e625cc1db52c
blob + 639f04b1cf41484047b231a54ed56b59103ff9c3
--- man/alpaca-config.5.scd
+++ man/alpaca-config.5.scd
*default_profile*
Profile to activate automatically when *--profile* is not given.
+*deepinfra_apikey*
+ API key for *alpaca quota -p deepinfra*. Unlike *apikey*, this applies
+ with no profile needed. Keep the file private if you set this value.
+
# PROFILES
A *[profile-name]* table defines a profile. Profiles inherit every top-level
*system*, *apikey*, *base_url* and *reasoning_effort*. A config setting
overrides a value saved in a transcript.
-*alpaca quota* uses *output_format*, *timeout*, *apikey* and *base_url*.
+*alpaca quota* uses *output_format* and *timeout*. Top-level *apikey* and
+*base_url* apply to Ollama only. Anthropic, OpenAI and DeepInfra use those
+settings only when a profile selected with *--profile* declares them. Values
+inherited from the top level, or from an automatic *default_profile*, do not
+replace their provider defaults. DeepInfra also accepts a key from top-level
+*deepinfra_apikey*, which needs no profile. An explicit profile's *apikey*
+still overrides it.
# EXAMPLE
blob - 1b9ef5568282a43dd6f351a1f94621076fe17e20
blob + 9862c8b8562daf49fff8cf13c67f42df4962f50a
--- man/alpaca-quota.1.scd
+++ man/alpaca-quota.1.scd
# DESCRIPTION
*alpaca quota* calls the usage endpoint of one provider with your credential.
-It prints each quota window as a bar showing the used share and a countdown to
-the next reset. Some providers also report a per-model request table.
+It prints each quota window as a bar showing the used share, with a countdown
+to the next reset. Some providers also report a per-model request table.
-The provider is chosen with *-p*, and defaults to *ollama*. Each provider has
-its own default credential and base URL, either of which can be overridden.
+Choose the provider with *-p*. It defaults to *ollama*. Each provider has its
+own default credential and base URL. You can override either.
*ollama*
- Ollama Cloud. Key from *OLLAMA_API_KEY*, base URL from
+ Ollama Cloud. Takes the key from *OLLAMA_API_KEY*, and the base URL from
*OLLAMA_API_ENDPOINT*, else *https://ollama.com*. Prints the session
- (5 hour) and weekly (7 day) quotas and a per-model request table for each.
- The endpoint reports no reset time, so the countdowns are computed locally:
- session windows align to UTC multiples of 5 hours from the epoch, and weekly
- windows use the same rule with a 4 day offset, so all accounts share the same
- boundary.
+ (5 hour) and weekly (7 day) quotas, each with a per-model request table.
+ The endpoint reports no reset time, so alpaca computes the countdowns
+ itself: session windows align to UTC multiples of 5 hours from the epoch,
+ and weekly windows use the same rule with a 4 day offset. This keeps the
+ boundary the same for every account.
*anthropic*
- Claude Code. Token from *CLAUDE_CODE_OAUTH_TOKEN*, else
+ Claude Code. Takes the token from *CLAUDE_CODE_OAUTH_TOKEN*, else
_~/.claude/.credentials.json_. Base URL *https://api.anthropic.com*.
- Prints whichever windows the account has, and extra spend when it is enabled.
+ Prints whichever windows the account has, plus extra spend when it is
+ enabled.
*openai*
- Codex CLI. Token from *CODEX_ACCESS_TOKEN*, else _~/.codex/auth.json_. Base
- URL *https://chatgpt.com*. Prints the plan, its rate limit windows, and the
- credit balance.
+ Codex CLI. Takes the token from *CODEX_ACCESS_TOKEN*, else
+ _~/.codex/auth.json_. Base URL *https://chatgpt.com*. Prints the plan, its
+ rate limit windows, and the credit balance.
-The *anthropic* and *openai* endpoints are not documented public APIs, so they
-may change without notice. Both need a short-lived OAuth token. If the
-credential file is stale, you will get an authentication error until the vendor
+*deepinfra*
+ DeepInfra. Takes the key from *DEEPINFRA_API_KEY*. Base URL
+ *https://api.deepinfra.com*. DeepInfra bills per token or per second, with
+ no fixed quota, so alpaca prints no window for it. Instead it prints the
+ current month's spend as a note, and billed units per model as a table.
+
+The *anthropic* and *openai* endpoints are not a documented public API, so
+they may change without notice. Both use a short-lived OAuth token. If the
+credential file goes stale, you get an authentication error until the vendor
CLI refreshes it.
# OPTIONS
*-p*, *--provider* _provider_
- One of *ollama*, *anthropic* or *openai*.
+ One of *ollama*, *anthropic*, *openai* or *deepinfra*.
*-T*, *--timeout* _secs_
Request timeout in seconds. Defaults to *ALPACA_TIMEOUT*, else 60.
# CONFIGURATION
-This command uses *output_format*, *timeout*, *apikey* and *base_url* from the
-Alpaca config file. Command-line options and environment variables take
+This command uses *output_format* and *timeout* from the Alpaca config file.
+Top-level *apikey* and *base_url* apply to Ollama only. For Anthropic, OpenAI
+or DeepInfra, declare connection values in a profile and select it with
+*--profile*. A profile chosen automatically through *default_profile* does not
+apply to those providers. Command-line options and environment variables take
priority. See *alpaca-config*(5).
# EXIT STATUS
blob - 56dfa03213b316112651c6e8ed43997494aaac7f
blob + b67830cd2c8ba83c8e309142a45e47cd594a0540
--- src/cli/config_tests.rs
+++ src/cli/config_tests.rs
}
}
+fn configured_quota(toml: &str, args: &[&str]) -> Result<QuotaArgs> {
+ let config = Config::from_toml(toml)?;
+ match parse_args_with_config(args, &config)? {
+ Invocation::Quota(args) => Ok(args),
+ other => Err(format!("expected quota invocation, got {other:?}").into()),
+ }
+}
+
#[test]
fn config_parses_every_supported_setting() -> Result<()> {
let args = configured_chat(
}
#[test]
+fn generic_connection_config_only_applies_to_ollama_quota() -> Result<()> {
+ let toml = r#"
+ apikey = "ollama-key"
+ base_url = "https://ollama.example"
+ "#;
+
+ let ollama = configured_quota(toml, &["alpaca", "quota"])?;
+ assert_eq!(ollama.api_key.as_deref(), Some("ollama-key"));
+ assert_eq!(ollama.base_url.as_deref(), Some("https://ollama.example"));
+
+ for provider in ["anthropic", "openai", "deepinfra"] {
+ let quota = configured_quota(toml, &["alpaca", "quota", "-p", provider])?;
+ assert_eq!(quota.api_key, None);
+ assert_eq!(quota.base_url, None);
+ }
+ Ok(())
+}
+
+#[test]
+fn deepinfra_apikey_is_used_without_a_profile() -> Result<()> {
+ let toml = r#"
+ apikey = "ollama-key"
+ deepinfra_apikey = "deepinfra-key"
+ "#;
+
+ let quota = configured_quota(toml, &["alpaca", "quota", "-p", "deepinfra"])?;
+ assert_eq!(quota.api_key.as_deref(), Some("deepinfra-key"));
+ assert_eq!(quota.base_url, None);
+
+ let ollama = configured_quota(toml, &["alpaca", "quota"])?;
+ assert_eq!(
+ ollama.api_key.as_deref(),
+ Some("ollama-key"),
+ "deepinfra_apikey does not leak into other providers"
+ );
+ Ok(())
+}
+
+#[test]
+fn an_explicit_profile_overrides_deepinfra_apikey() -> Result<()> {
+ let toml = r#"
+ deepinfra_apikey = "top-level-key"
+
+ [work]
+ apikey = "profile-key"
+ "#;
+
+ let quota = configured_quota(
+ toml,
+ &["alpaca", "quota", "--profile", "work", "-p", "deepinfra"],
+ )?;
+ assert_eq!(quota.api_key.as_deref(), Some("profile-key"));
+ Ok(())
+}
+
+#[test]
+fn non_ollama_quota_only_uses_connection_values_from_explicit_profile() -> Result<()> {
+ let toml = r#"
+ apikey = "ollama-key"
+ base_url = "https://ollama.example"
+ default_profile = "ollama-cloud"
+
+ [ollama-cloud]
+ apikey = "cloud-key"
+ base_url = "https://cloud.example"
+ timeout = 17
+
+ [codex]
+ apikey = "codex-key"
+ base_url = "https://codex.example"
+ "#;
+
+ let automatic = configured_quota(toml, &["alpaca", "quota", "-p", "openai"])?;
+ assert_eq!(automatic.api_key, None);
+ assert_eq!(automatic.base_url, None);
+ assert_eq!(automatic.timeout, Duration::from_secs(17));
+
+ let explicit = configured_quota(
+ toml,
+ &["alpaca", "quota", "--profile", "codex", "-p", "openai"],
+ )?;
+ assert_eq!(explicit.api_key.as_deref(), Some("codex-key"));
+ assert_eq!(explicit.base_url.as_deref(), Some("https://codex.example"));
+ Ok(())
+}
+
+#[test]
fn config_rejects_unknown_keys_and_non_positive_timeout() {
assert!(Config::from_toml("modle = 'typo'").is_err());
assert!(Config::from_toml("timeout = 0").is_err());
blob - d2e1fba388bd246eb53dc21bc31e52e2e00efa22
blob + c8fc471830becd0924c55338e11e820ce9349999
--- src/cli.rs
+++ src/cli.rs
base_url: Option<String>,
reasoning_effort: Option<Effort>,
default_profile: Option<String>,
+ deepinfra_apikey: Option<String>,
}
#[derive(Debug, Default, Clone)]
profiles: HashMap::new(),
})
}
+
+ /// Resolve an active profile and prevent generic Ollama connection
+ /// settings from being reused by another quota provider.
+ fn for_invocation(&self, profile: Option<&str>, provider: Option<Provider>) -> Result<Self> {
+ let active = profile.or(self.default_profile.as_deref());
+ let mut config = match active {
+ Some(name) => self.with_profile(name)?,
+ None => self.clone(),
+ };
+
+ if matches!(
+ provider,
+ Some(Provider::Anthropic | Provider::OpenAI | Provider::DeepInfra)
+ ) {
+ let explicit = profile.and_then(|name| self.profiles.get(name));
+ config.settings.apikey = explicit.and_then(|settings| settings.apikey.clone());
+ config.settings.base_url = explicit.and_then(|settings| settings.base_url.clone());
+ }
+
+ // deepinfra_apikey is a top-level fallback, so it applies even without
+ // an explicit profile.
+ if provider == Some(Provider::DeepInfra) {
+ config.settings.apikey = config
+ .settings
+ .apikey
+ .clone()
+ .or_else(|| self.settings.deepinfra_apikey.clone());
+ }
+
+ Ok(config)
+ }
}
impl Settings {
.default_profile
.clone()
.or_else(|| self.default_profile.clone()),
+ deepinfra_apikey: other
+ .deepinfra_apikey
+ .clone()
+ .or_else(|| self.deepinfra_apikey.clone()),
}
}
}
.try_get_matches_from(&args)
.unwrap_or_else(|err| err.exit());
let profile = pre_matches.get_one::<String>("profile").cloned();
+ let provider = quota_provider(&pre_matches);
let config = Config::load()?;
- let profile = profile.or_else(|| config.default_profile.clone());
- let config = match profile {
- Some(name) => config.with_profile(&name)?,
- None => config,
- };
+ let config = config.for_invocation(profile.as_deref(), provider)?;
let matches = cli(&config).get_matches_from(args);
Ok(invocation_from_matches(&matches, &config))
fn parse_args_with_config(args: &[&str], config: &Config) -> Result<Invocation> {
let pre_matches = cli(config).try_get_matches_from(args)?;
let profile = pre_matches.get_one::<String>("profile").cloned();
- let profile = profile.or_else(|| config.default_profile.clone());
- let config = match profile {
- Some(name) => config.with_profile(&name)?,
- None => config.clone(),
- };
+ let provider = quota_provider(&pre_matches);
+ let config = config.for_invocation(profile.as_deref(), provider)?;
let matches = cli(&config).try_get_matches_from(args)?;
Ok(invocation_from_matches(&matches, &config))
}
+fn quota_provider(matches: &ArgMatches) -> Option<Provider> {
+ matches
+ .subcommand_matches("quota")
+ .and_then(|quota| quota.get_one::<Provider>("provider"))
+ .copied()
+}
+
/// Arguments shared by the default command and `alpaca chat`
fn chat_args(cmd: Command, config: &Config) -> Command {
output_format_args(cmd, config)
fn quota_command(config: &Config) -> Command {
output_format_args(
- Command::new("quota").about("Report how much provider quota is used"),
+ Command::new("quota")
+ .visible_alias("q")
+ .about("Report how much provider quota is used"),
config,
)
.arg(
fn compose_command(config: &Config) -> Command {
output_format_args(
- Command::new("compose").about("Edit a prompt in $EDITOR, then send it"),
+ Command::new("compose")
+ .visible_alias("c")
+ .about("Edit a prompt in $EDITOR, then send it"),
config,
)
.arg(default_value_opt(model_arg(), config.model.clone()))
impl ValueEnum for Provider {
fn value_variants<'a>() -> &'a [Self] {
- &[Self::Ollama, Self::Anthropic, Self::OpenAI]
+ &[Self::Ollama, Self::Anthropic, Self::OpenAI, Self::DeepInfra]
}
fn to_possible_value(&self) -> Option<PossibleValue> {
Self::Ollama => "ollama",
Self::Anthropic => "anthropic",
Self::OpenAI => "openai",
+ Self::DeepInfra => "deepinfra",
}))
}
}
}
#[test]
+ fn quota_alias_q() -> Result<()> {
+ let args = quota_args(&["alpaca", "q"])?;
+ assert_eq!(args.provider, Provider::Ollama);
+ Ok(())
+ }
+
+ #[test]
+ fn compose_alias_c() -> Result<()> {
+ let args = compose_args(&["alpaca", "c"])?;
+ assert_eq!(args.model, None);
+ Ok(())
+ }
+
+ #[test]
fn quota_json_shorthand() -> Result<()> {
let args = quota_args(&["alpaca", "quota", "--json"])?;
assert_eq!(args.output_format, OutputFormat::JSON);
blob - 858707aa6886d8401f46bafc36d226f20f923c4f
blob + ca996b9323cfcf11e3b73e8117ef57362229a2c3
--- src/usage/mod.rs
+++ src/usage/mod.rs
//! about quota windows and per-model request counts.
pub mod anthropic;
+pub mod deepinfra;
pub mod ollama;
pub mod openai;
pub mod render;
use crate::Error;
+struct Credential {
+ token: String,
+ account_id: Option<String>,
+}
+
/// Largest response body read from a usage endpoint
const MAX_BODY: usize = 1 << 20;
Ollama,
Anthropic,
OpenAI,
+ DeepInfra,
}
/// Provider usage, normalised for rendering
pub resets_in_secs: Option<u64>,
}
-/// One table of per-model request counts
+/// One table of per-model counts
#[derive(Debug, Clone, PartialEq, Serialize)]
pub struct Table {
pub heading: String,
+ /// Name of the counted unit, shown as the column header (e.g. "reqs")
+ pub unit: &'static str,
pub rows: Vec<ModelCount>,
}
-/// Requests made against one model
+/// A count of some unit against one model, named by the enclosing [`Table`]
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct ModelCount {
pub name: String,
.unwrap_or_else(|| "https://ollama.com".to_string()),
Self::Anthropic => "https://api.anthropic.com".to_string(),
Self::OpenAI => "https://chatgpt.com".to_string(),
+ Self::DeepInfra => "https://api.deepinfra.com".to_string(),
}
}
Self::Ollama => "/api/usage",
Self::Anthropic => "/api/oauth/usage",
Self::OpenAI => "/backend-api/codex/usage",
+ // `from=current` selects the current billing month.
+ Self::DeepInfra => "/payment/usage?from=current",
}
}
/// Whether the credential is sent as a bearer token
#[must_use]
pub fn bearer(self) -> bool {
- matches!(self, Self::Anthropic | Self::OpenAI)
+ matches!(self, Self::Anthropic | Self::OpenAI | Self::DeepInfra)
}
/// The credential this provider uses when none is given on the command line
///
/// Returns [`Error::NoCredentials`] when no environment variable is set and
/// the vendor CLI's credential file is missing or malformed.
- pub fn credential(self) -> Result<String, Error> {
+ fn credential(self) -> Result<Credential, Error> {
let from_env = |name: &str| std::env::var(name).ok().filter(|key| !key.is_empty());
- let key = match self {
- Self::Ollama => from_env("OLLAMA_API_KEY"),
+ let credential = match self {
+ Self::Ollama => from_env("OLLAMA_API_KEY").map(|token| Credential {
+ token,
+ account_id: None,
+ }),
Self::Anthropic => from_env("CLAUDE_CODE_OAUTH_TOKEN")
- .or_else(|| anthropic::token_from_file().ok().flatten()),
- Self::OpenAI => {
- from_env("CODEX_ACCESS_TOKEN").or_else(|| openai::token_from_file().ok().flatten())
- }
+ .or_else(|| anthropic::token_from_file().ok().flatten())
+ .map(|token| Credential {
+ token,
+ account_id: None,
+ }),
+ Self::OpenAI => from_env("CODEX_ACCESS_TOKEN")
+ .map(|token| Credential {
+ token,
+ account_id: None,
+ })
+ .or_else(|| {
+ openai::credential_from_file()
+ .ok()
+ .flatten()
+ .map(|(token, account_id)| Credential { token, account_id })
+ }),
+ Self::DeepInfra => from_env("DEEPINFRA_API_KEY").map(|token| Credential {
+ token,
+ account_id: None,
+ }),
};
- key.ok_or(Error::NoCredentials { provider: self })
+ credential.ok_or(Error::NoCredentials { provider: self })
}
/// Normalise a usage response body
Self::Ollama => ollama::view(body),
Self::Anthropic => anthropic::view(body),
Self::OpenAI => openai::view(body),
+ Self::DeepInfra => deepinfra::view(body),
}
}
/// over TLS 1.2. No compression feature is enabled for reqwest, so no
/// `Accept-Encoding` header is sent either, which the same filter requires.
fn http_client(self) -> Result<reqwest::Client, Error> {
- let builder = reqwest::Client::builder().user_agent(user_agent());
+ let agent = match self {
+ Self::Anthropic => "claude-cli (external, cli)".to_string(),
+ _ => user_agent(),
+ };
+ let builder = reqwest::Client::builder().user_agent(agent);
let builder = match self {
Self::OpenAI => builder
.http1_only()
Self::Ollama => "ollama",
Self::Anthropic => "anthropic",
Self::OpenAI => "openai",
+ Self::DeepInfra => "deepinfra",
};
f.write_str(name)
}
provider: Provider,
base_url: String,
api_key: String,
+ account_id: Option<String>,
timeout: Duration,
}
base_url: Option<String>,
timeout: Duration,
) -> Result<Self, Error> {
- let api_key = match api_key {
- Some(key) => key,
+ let credential = match api_key {
+ Some(token) => Credential {
+ token,
+ account_id: None,
+ },
None => provider.credential()?,
};
http: provider.http_client()?,
provider,
base_url: base_url.unwrap_or_else(|| provider.base_url()),
- api_key,
+ api_key: credential.token,
+ account_id: credential.account_id,
timeout,
})
}
self.api_key.clone()
};
- let resp = self
+ let mut request = self
.http
.get(self.endpoint())
.timeout(self.timeout)
.header("Authorization", credential)
- .header("Accept", "application/json")
- .send()
- .await?;
+ .header("Accept", "application/json");
+ if self.provider == Provider::Anthropic {
+ request = request.header("anthropic-beta", "oauth-2025-04-20");
+ }
+ if let Some(account_id) = &self.account_id {
+ request = request.header("ChatGPT-Account-Id", account_id);
+ }
+
+ let resp = request.send().await?;
+
let status = resp.status();
let body = read_body(resp).await?;
assert_eq!(Provider::Ollama.to_string(), "ollama");
assert_eq!(Provider::Anthropic.to_string(), "anthropic");
assert_eq!(Provider::OpenAI.to_string(), "openai");
+ assert_eq!(Provider::DeepInfra.to_string(), "deepinfra");
}
#[test]
assert!(!Provider::Ollama.bearer());
assert!(Provider::Anthropic.bearer());
assert!(Provider::OpenAI.bearer());
+ assert!(Provider::DeepInfra.bearer());
}
#[test]
blob - /dev/null
blob + 9340a12fb201054eeaee820e6b34cab833858ce3 (mode 644)
--- /dev/null
+++ src/usage/deepinfra.rs
+//! `DeepInfra` usage
+//!
+//! `DeepInfra` bills per token or per second with no fixed quota, so there is no
+//! window to show a used share of. The billing endpoint instead reports the
+//! current month's spend, which becomes the trailing note, and a per-model
+//! breakdown of billed units, mirroring Ollama's per-model request table.
+
+use serde::Deserialize;
+
+use super::{ModelCount, Table, View};
+use crate::Error;
+
+#[derive(Debug, Default, Deserialize)]
+struct ModelMeta {
+ #[serde(default)]
+ model_name: String,
+}
+
+#[derive(Debug, Default, Deserialize)]
+struct UsageItem {
+ #[serde(default)]
+ model: ModelMeta,
+ /// Billed seconds or tokens
+ #[serde(default)]
+ units: u64,
+}
+
+#[derive(Debug, Default, Deserialize)]
+struct UsageMonth {
+ #[serde(default)]
+ period: String,
+ /// Total cost for the period, in cents
+ #[serde(default)]
+ total_cost: i64,
+ #[serde(default)]
+ items: Vec<UsageItem>,
+}
+
+#[derive(Debug, Deserialize)]
+struct Usage {
+ #[serde(default)]
+ months: Vec<UsageMonth>,
+}
+
+/// Normalise the `DeepInfra` billing usage response
+///
+/// # Errors
+///
+/// Returns an error if the body is not a `DeepInfra` usage response.
+pub fn view(body: &[u8]) -> Result<View, Error> {
+ let value: serde_json::Value =
+ serde_json::from_slice(body).map_err(|err| Error::UnexpectedResponse(err.to_string()))?;
+ if !["months", "initial_month"]
+ .iter()
+ .any(|key| value.get(key).is_some())
+ {
+ return Err(Error::UnexpectedResponse(
+ "response contains no deepinfra usage fields".to_string(),
+ ));
+ }
+ let usage: Usage =
+ serde_json::from_value(value).map_err(|err| Error::UnexpectedResponse(err.to_string()))?;
+
+ let month = usage.months.first();
+ let items = month.map_or([].as_slice(), |month| month.items.as_slice());
+
+ Ok(View {
+ title: "deepinfra usage".to_string(),
+ subtitle: month
+ .map(|month| month.period.clone())
+ .filter(|period| !period.is_empty()),
+ note: month.map(|month| format!("total {}", cents(month.total_cost))),
+ windows: vec![],
+ tables: vec![Table {
+ heading: "models".to_string(),
+ unit: "units",
+ rows: by_model(items),
+ }],
+ })
+}
+
+/// Format a cent amount as a dollar figure
+fn cents(cents: i64) -> String {
+ let whole = cents / 100;
+ let fraction = cents.abs() % 100;
+ format!("{whole}.{fraction:02} USD")
+}
+
+/// Billed units per model, busiest first
+///
+/// A model can appear in several items when its pricing changed mid-month, so
+/// units are summed per model name before sorting.
+fn by_model(items: &[UsageItem]) -> Vec<ModelCount> {
+ let mut totals: Vec<ModelCount> = Vec::new();
+ for item in items {
+ if item.model.model_name.is_empty() {
+ continue;
+ }
+ match totals
+ .iter_mut()
+ .find(|count| count.name == item.model.model_name)
+ {
+ Some(count) => count.requests += item.units,
+ None => totals.push(ModelCount {
+ name: item.model.model_name.clone(),
+ requests: item.units,
+ }),
+ }
+ }
+ totals.sort_by_key(|count| std::cmp::Reverse(count.requests));
+ totals
+}
+
+#[cfg(test)]
+mod test {
+ use super::*;
+ use anyhow::Result;
+
+ #[test]
+ fn reads_the_current_months_total_cost_and_models() -> Result<()> {
+ let body = br#"{
+ "months": [{"period": "2026.09", "interval": {}, "total_cost": 1234, "items": [
+ {"model": {"model_name": "meta-llama/Llama-3.1-70B"}, "units": 100, "rate": 0, "cost": 900, "pricing_type": "token", "interval": {}},
+ {"model": {"model_name": "meta-llama/Llama-3.1-70B"}, "units": 50, "rate": 0, "cost": 100, "pricing_type": "token", "interval": {}},
+ {"model": {"model_name": "Qwen/Qwen2.5-7B"}, "units": 80, "rate": 0, "cost": 234, "pricing_type": "token", "interval": {}}
+ ]}],
+ "initial_month": "2024.01"
+ }"#;
+
+ let view = view(body)?;
+
+ assert_eq!(view.title, "deepinfra usage");
+ assert_eq!(view.subtitle.as_deref(), Some("2026.09"));
+ assert_eq!(view.note.as_deref(), Some("total 12.34 USD"));
+ assert!(view.windows.is_empty());
+ assert_eq!(view.tables[0].heading, "models");
+ assert_eq!(view.tables[0].unit, "units");
+ assert_eq!(
+ view.tables[0].rows,
+ vec![
+ ModelCount {
+ name: "meta-llama/Llama-3.1-70B".to_string(),
+ requests: 150
+ },
+ ModelCount {
+ name: "Qwen/Qwen2.5-7B".to_string(),
+ requests: 80
+ }
+ ],
+ "units for the same model are summed, then sorted busiest first"
+ );
+ Ok(())
+ }
+
+ #[test]
+ fn omits_note_and_subtitle_without_months() -> Result<()> {
+ let view = view(br#"{"months": [], "initial_month": "2024.01"}"#)?;
+
+ assert_eq!(view.note, None);
+ assert_eq!(view.subtitle, None);
+ assert!(view.tables[0].rows.is_empty());
+ Ok(())
+ }
+
+ #[test]
+ fn rejects_non_usage_bodies() {
+ assert!(view(b"not json").is_err());
+ assert!(view(b"{}").is_err());
+ }
+}
blob - f430e5638ccc6d2517cdc38eed5d1a76ca3d56f2
blob + 7edd3c721a6b7ae084eb5b2dc8b9b97b1bb07917
--- src/usage/ollama.rs
+++ src/usage/ollama.rs
tables: vec![
Table {
heading: "session models".to_string(),
+ unit: "reqs",
rows: sorted(&usage.limits.session.models),
},
Table {
heading: "weekly models".to_string(),
+ unit: "reqs",
rows: sorted(&usage.limits.weekly.models),
},
],
blob - af79856fd6db58534020742a9dcedcf3f79bdd04
blob + 416d9d517bd6246ab07f8d5dd6d96a468b3fc895
--- src/usage/openai.rs
+++ src/usage/openai.rs
credits: Credits,
}
-/// The OAuth token the Codex CLI writes when you log in
+/// The OAuth token and account ID the Codex CLI writes when you log in
///
/// # Errors
///
/// Returns an error if the credential file exists but cannot be parsed.
-pub fn token_from_file() -> Result<Option<String>, Error> {
+pub fn credential_from_file() -> Result<Option<(String, Option<String>)>, Error> {
#[derive(Deserialize)]
struct Auth {
tokens: Tokens,
#[derive(Deserialize)]
struct Tokens {
access_token: String,
+ #[serde(default)]
+ account_id: Option<String>,
}
let path = crate::home_dir()?.join(".codex").join("auth.json");
};
let auth: Auth = serde_json::from_slice(&data)?;
- Ok(Some(auth.tokens.access_token).filter(|token| !token.is_empty()))
+ let token = auth.tokens.access_token;
+ let account_id = auth.tokens.account_id.filter(|id| !id.is_empty());
+ Ok((!token.is_empty()).then_some((token, account_id)))
}
/// Normalise the Codex usage response
blob - 22ae73796ea44d2c93c71930324702ae2e3b28ee
blob + 9792cdefe7306a63fd9505a433b040f9ac390aed
--- src/usage/render.rs
+++ src/usage/render.rs
let headings = view
.tables
.iter()
- .map(|table| format!("{:<NAME_WIDTH$} {:>COUNT_WIDTH$}", table.heading, "reqs"))
+ .map(|table| {
+ format!(
+ "{:<NAME_WIDTH$} {:>COUNT_WIDTH$}",
+ table.heading, table.unit
+ )
+ })
.collect::<Vec<_>>()
.join(" ");
writeln!(
],
tables: vec![Table {
heading: "session models".to_string(),
+ unit: "reqs",
rows: vec![ModelCount {
name: "big".to_string(),
requests: 9,
let view = View {
tables: vec![Table {
heading: "models".to_string(),
+ unit: "reqs",
rows: vec![],
}],
..view()
blob - 1cd9a68a621bd35fa406dfaa708eeb65132392e0
blob + c0c5d2d74dcc99af46eed8d8e2bda1cea623dc32
--- tests/chat.rs
+++ tests/chat.rs
use assert_fs::prelude::*;
use predicates::prelude::*;
use serde_json::json;
+use std::sync::LazyLock;
+use tempfile::TempDir;
+static ISOLATED_CONFIG_HOME: LazyLock<TempDir> =
+ LazyLock::new(|| TempDir::new().expect("failed to create isolated config dir"));
+
+fn alpaca() -> Command {
+ let mut cmd = Command::cargo_bin("alpaca").unwrap();
+ cmd.env("XDG_CONFIG_HOME", ISOLATED_CONFIG_HOME.path());
+ cmd
+}
+
/// A canned successful native `/api/chat` response body
fn ok_body() -> &'static str {
r#"{
#[test]
fn chat_no_message() {
- Command::cargo_bin("alpaca")
- .unwrap()
+ alpaca()
.assert()
.success()
.stdout(predicate::str::contains("Usage: alpaca"));
#[test]
fn chat_no_file() {
- Command::cargo_bin("alpaca")
- .unwrap()
+ alpaca()
.args(["file_does_not_exist"])
.assert()
.failure()
.with_body(ok_body())
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args(["-u", "Hello"])
.env("API_ENDPOINT", server.url())
.env("API_KEY", "ABCDE")
.with_body(ok_body())
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args(["-"])
.write_stdin("Hello")
.env("API_ENDPOINT", server.url())
.with_body(ok_body())
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args(["-u", "Hello", "--reasoning-effort", "medium"])
.env("API_ENDPOINT", server.url())
.env("API_KEY", "ABCDE")
.with_body(ok_body())
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args([
"-s", "SYSTEM", "-u", "USER_1", "-a", "ASSI_1", "-u", "USER_2", "-a", "ASSI_2",
])
.with_body(r#"{ "error": "invalid options: temperature out of range" }"#)
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args(["-u", "USER", "-t", "1000"])
.write_stdin("USER_STDIN")
.env("API_ENDPOINT", server.url())
)
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args([
"-u",
"Hello",
.with_body(ok_body())
.create();
- let cmd = Command::cargo_bin("alpaca")
- .unwrap()
+ let cmd = alpaca()
.args([infile.path().to_str().unwrap()])
.env("API_ENDPOINT", server.url())
.env("API_KEY", "ABCDE")
blob - 9736052666c25a9bc0b82b75fc7c01fbd428aec4
blob + 4b686f032d9b144b55be19d40cf878adfedcc433
--- tests/quota.rs
+++ tests/quota.rs
use assert_cmd::Command;
use predicates::prelude::*;
+use std::sync::LazyLock;
+use tempfile::TempDir;
+static ISOLATED_CONFIG_HOME: LazyLock<TempDir> =
+ LazyLock::new(|| TempDir::new().expect("failed to create isolated config dir"));
+
/// A canned successful Ollama `/api/usage` response body
fn ok_body() -> &'static str {
r#"{
}
fn alpaca() -> Command {
- Command::cargo_bin("alpaca").unwrap()
+ let mut cmd = Command::cargo_bin("alpaca").unwrap();
+ cmd.env("XDG_CONFIG_HOME", ISOLATED_CONFIG_HOME.path());
+ cmd
}
#[test]
"response body exceeds 1048576 bytes",
));
}
+
+#[test]
+fn anthropic_quota_sends_oauth_beta_header() {
+ let mut server = mockito::Server::new();
+ let mock = server
+ .mock("GET", "/api/oauth/usage")
+ .match_header("authorization", "Bearer TOKEN")
+ .match_header("anthropic-beta", "oauth-2025-04-20")
+ .match_header("user-agent", "claude-cli (external, cli)")
+ .with_body(r#"{"five_hour":{"utilization":25}}"#)
+ .create();
+
+ alpaca()
+ .args([
+ "quota",
+ "-p",
+ "anthropic",
+ "--apikey",
+ "TOKEN",
+ "--base-url",
+ &server.url(),
+ ])
+ .assert()
+ .success();
+
+ mock.assert();
+}
+
+#[test]
+fn deepinfra_quota_reports_the_current_months_spend() {
+ let mut server = mockito::Server::new();
+ let mock = server
+ .mock("GET", "/payment/usage?from=current")
+ .match_header("authorization", "Bearer KEY")
+ .with_body(
+ r#"{"months": [{"period": "2026.09", "interval": {}, "total_cost": 1234, "items": [
+ {"model": {"model_name": "meta-llama/Llama-3.1-70B"}, "units": 100, "rate": 0, "cost": 1234, "pricing_type": "token", "interval": {}}
+ ]}],
+ "initial_month": "2024.01"}"#,
+ )
+ .create();
+
+ alpaca()
+ .args([
+ "quota",
+ "-p",
+ "deepinfra",
+ "--apikey",
+ "KEY",
+ "--base-url",
+ &server.url(),
+ ])
+ .assert()
+ .success()
+ .stdout(
+ predicate::str::contains("deepinfra usage")
+ .and(predicate::str::contains("2026.09"))
+ .and(predicate::str::contains("total 12.34 USD"))
+ .and(predicate::str::contains("meta-llama/Llama-3.1-70B"))
+ .and(predicate::str::contains("units")),
+ );
+
+ mock.assert();
+}
+
+#[test]
+fn openai_quota_sends_account_id_from_codex_credentials() {
+ let home = TempDir::new().unwrap();
+ let codex = home.path().join(".codex");
+ std::fs::create_dir(&codex).unwrap();
+ std::fs::write(
+ codex.join("auth.json"),
+ r#"{"tokens":{"access_token":"TOKEN","account_id":"ACCOUNT"}}"#,
+ )
+ .unwrap();
+
+ let mut server = mockito::Server::new();
+ let mock = server
+ .mock("GET", "/backend-api/codex/usage")
+ .match_header("authorization", "Bearer TOKEN")
+ .match_header("chatgpt-account-id", "ACCOUNT")
+ .with_body(r#"{"plan_type":"plus"}"#)
+ .create();
+
+ alpaca()
+ .env("HOME", home.path())
+ .env_remove("CODEX_ACCESS_TOKEN")
+ .args(["quota", "-p", "openai", "--base-url", &server.url()])
+ .assert()
+ .success();
+
+ mock.assert();
+}