2424//
2525// A miss is benign: the failure stays on the existing classification.
2626
27+ import { parseChallenges } from "./www-authenticate" ;
28+
2729export type InsufficientScopeDetection = {
2830 /** Scopes the upstream named as required, when it named any (RFC 6750's
2931 * `scope` attribute). Empty when the provider only signalled the class of
@@ -36,172 +38,6 @@ const MAX_DEPTH = 8;
3638const isRecord = ( value : unknown ) : value is Record < string , unknown > =>
3739 typeof value === "object" && value !== null && ! Array . isArray ( value ) ;
3840
39- /** Parser for the whole WWW-Authenticate header per RFC 7235 §2.1: a
40- * comma-separated #list of challenges, each
41- * `scheme [ 1*SP ( token68 / #auth-param ) ]`. Implemented as an explicit
42- * per-challenge state machine so params can never attach across challenge
43- * boundaries or to a token68 credential:
44- *
45- * - "scheme": just read a scheme; accepts a token68 OR a first auth-param
46- * (space-separated, no comma).
47- * - "params": accepts further auth-params ONLY after a comma.
48- * - "token68": accepts nothing; any trailing param is malformed.
49- *
50- * Auth-params allow BWS around `=` (RFC 7230). Quoted-strings consume
51- * quoted-pairs whole and must end at a separator. ANY malformed shape —
52- * scheme-less params, space-separated param runs, params after token68,
53- * stray quotes/bytes — returns null and never classifies: a miss is benign,
54- * a false positive strips a valid recovery path. */
55- type Challenge = { readonly scheme : string ; readonly params : Map < string , string > } ;
56-
57- // HTTP `token` alphabet (RFC 7230 §3.2.6) — schemes and auth-param names.
58- const TOKEN_RE = / [ A - Z a - z 0 - 9 ! # $ % & ' * + . ^ _ ` | ~ - ] / ;
59- // token68 alphabet (RFC 7235 §2.1), padding `=` handled separately.
60- const TOKEN68_RE = / [ A - Z a - z 0 - 9 . _ ~ + / - ] / ;
61- // Superset used by the word reader; each use site validates against the
62- // context-specific alphabet after reading.
63- const WORD_RE = / [ A - Z a - z 0 - 9 ! # $ % & ' * + . ^ _ ` | ~ / - ] / ;
64-
65- const isToken = ( word : string ) : boolean => [ ...word ] . every ( ( ch ) => TOKEN_RE . test ( ch ) ) ;
66- // Unquoted URL values some providers emit (scheme://host/path?query): URI
67- // characters per RFC 3986, no whitespace/comma/quotes.
68- const isUrlish = ( word : string ) : boolean => / ^ [ A - Z a - z ] [ A - Z a - z 0 - 9 + . - ] * : \/ \/ [ ^ \s , " ] + $ / . test ( word ) ;
69- const isToken68 = ( word : string ) : boolean => [ ...word ] . every ( ( ch ) => TOKEN68_RE . test ( ch ) ) ;
70-
71- const parseChallenges = ( header : string ) : readonly Challenge [ ] | null => {
72- const len = header . length ;
73- const challenges : Challenge [ ] = [ ] ;
74- let current : Challenge | null = null ;
75- let state : "boundary" | "scheme" | "token68" | "params" = "boundary" ;
76- let sawComma = true ; // header start counts as a list boundary
77- let i = 0 ;
78-
79- const readWord = ( ) : string => {
80- const start = i ;
81- while ( i < len && WORD_RE . test ( header [ i ] ! ) ) i += 1 ;
82- return header . slice ( start , i ) ;
83- } ;
84- // Returns null on an unterminated quote or a quote run into the next token.
85- const readQuoted = ( ) : string | null => {
86- let value = "" ;
87- i += 1 ; // opening quote
88- while ( i < len ) {
89- const ch = header [ i ] ! ;
90- if ( ch === '"' ) {
91- i += 1 ;
92- return i >= len || / [ \s , ] / . test ( header [ i ] ! ) ? value : null ;
93- }
94- if ( ch === "\\" && i + 1 < len ) {
95- value += header [ i + 1 ] ;
96- i += 2 ;
97- continue ;
98- }
99- value += ch ;
100- i += 1 ;
101- }
102- return null ; // unterminated
103- } ;
104-
105- while ( i < len ) {
106- while ( i < len && / \s / . test ( header [ i ] ! ) ) i += 1 ;
107- if ( i >= len ) break ;
108- if ( header [ i ] === "," ) {
109- sawComma = true ;
110- i += 1 ;
111- continue ;
112- }
113- if ( ! WORD_RE . test ( header [ i ] ! ) ) return null ; // stray quote/byte: malformed
114- const word = readWord ( ) ;
115- // Look ahead through BWS for `=` to classify the word.
116- let j = i ;
117- while ( j < len && / [ \t ] / . test ( header [ j ] ! ) ) j += 1 ;
118- const isPaddingRun = ( ( ) => {
119- // An `=`-run directly on the word (no BWS) that is followed (after
120- // optional whitespace) by a comma or the end of input is token68
121- // padding. An `=` followed by a value — even across BWS — is an
122- // auth-param (RFC 7230 allows BWS around `=`).
123- if ( header [ i ] !== "=" ) return false ;
124- let k = i ;
125- while ( k < len && header [ k ] === "=" ) k += 1 ;
126- while ( k < len && / [ \t ] / . test ( header [ k ] ! ) ) k += 1 ;
127- return k >= len || header [ k ] === "," ;
128- } ) ( ) ;
129-
130- if ( isPaddingRun ) {
131- // token68 with padding — only legal directly after a scheme.
132- if ( state !== "scheme" || sawComma ) return null ;
133- if ( ! isToken68 ( word ) ) return null ;
134- while ( i < len && header [ i ] === "=" ) i += 1 ;
135- state = "token68" ;
136- sawComma = false ;
137- continue ;
138- }
139-
140- if ( header [ j ] === "=" ) {
141- // auth-param: `word BWS = BWS value`.
142- if ( ! isToken ( word ) ) return null ; // param name must be an HTTP token
143- if ( current === null ) return null ; // scheme-less param
144- if ( state === "token68" ) return null ; // params after token68
145- if ( state === "scheme" && sawComma ) return null ; // "Bearer, a=b"
146- if ( state === "params" && ! sawComma ) return null ; // space-separated run
147- i = j + 1 ;
148- while ( i < len && / [ \t ] / . test ( header [ i ] ! ) ) i += 1 ;
149- let value : string ;
150- if ( header [ i ] === '"' ) {
151- const quoted = readQuoted ( ) ;
152- if ( quoted === null ) return null ;
153- value = quoted ;
154- } else {
155- const start = i ;
156- while ( i < len && ! / [ \s , ] / . test ( header [ i ] ! ) ) i += 1 ;
157- value = header . slice ( start , i ) ;
158- // An unquoted value must be an HTTP token (`realm =,` / `realm=;`
159- // are malformed) — EXCEPT that real providers emit unquoted URLs for
160- // resource_metadata (observed live: Stripe), so URL-safe characters
161- // are tolerated there. The signal params (`error`, `scope`) stay
162- // token-strict.
163- if ( value . length === 0 ) return null ;
164- const lowerName = word . toLowerCase ( ) ;
165- if ( ! isToken ( value ) && ! ( lowerName === "resource_metadata" && isUrlish ( value ) ) ) {
166- return null ;
167- }
168- }
169- // Duplicate SIGNAL params (`error`, `scope`) within one challenge mean
170- // a header playing games — never classify. Other duplicates are
171- // tolerated first-wins: real providers emit them (observed live:
172- // Sentry duplicates resource_metadata).
173- const key = word . toLowerCase ( ) ;
174- if ( current . params . has ( key ) ) {
175- if ( key === "error" || key === "scope" ) return null ;
176- } else {
177- current . params . set ( key , value ) ;
178- }
179- state = "params" ;
180- sawComma = false ;
181- continue ;
182- }
183-
184- // Bare word: a new challenge's scheme at a list boundary, a token68
185- // directly after a scheme, malformed anywhere else.
186- if ( sawComma ) {
187- if ( ! isToken ( word ) ) return null ; // a scheme must be an HTTP token
188- current = { scheme : word . toLowerCase ( ) , params : new Map ( ) } ;
189- challenges . push ( current ) ;
190- state = "scheme" ;
191- sawComma = false ;
192- continue ;
193- }
194- if ( state === "scheme" ) {
195- if ( ! isToken68 ( word ) ) return null ;
196- state = "token68" ;
197- continue ;
198- }
199- return null ;
200- }
201-
202- return challenges ;
203- } ;
204-
20541const detectFromChallenge = ( header : string ) : InsufficientScopeDetection | null => {
20642 const challenges = parseChallenges ( header ) ;
20743 if ( challenges === null ) return null ;
0 commit comments