; oligo-analysis  -v 1 -sort -i $RSAT/public_html/tmp/www-data/2026/06/02/tmp_sequence_2026-06-02.214645_y0sL7m.fasta.purged -format fasta -lth occ_sig 0 -uth rank 50 -return occ,proba,rank -2str -noov -quick_if_possible -seqtype dna -bg upstream-noorf -org Homo_sapiens_GRCh38 -pseudo 0.01 -l 6 -o $RSAT/public_html/tmp/www-data/2026/06/02/oligo-analysis_2026-06-02.214645_NYFFii_6nt.tab
; Citation: van Helden et al. (1998). J Mol Biol 281(5), 827-42. 
; Program version              	1.169
; Quick counting mode          
; Detection of over-represented words (right-tail test)
; Oligomer length              	6
; Input file                   	$RSAT/public_html/tmp/www-data/2026/06/02/tmp_sequence_2026-06-02.214645_y0sL7m.fasta.purged
; Input format                 	fasta
; Output file                  	$RSAT/public_html/tmp/www-data/2026/06/02/oligo-analysis_2026-06-02.214645_NYFFii_6nt.tab
; Discard overlapping matches
; Counted on both strands
; 	grouped by pairs of reverse complements
; Background model             	upstream-noorf
; Organism                     	Homo_sapiens_GRCh38
; Background estimation method 	Frequency file
; Expected frequency file      	$RSAT/public_html/data/genomes/Homo_sapiens_GRCh38/oligo-frequencies/6nt_upstream-noorf_Homo_sapiens_GRCh38-noov-2str.freq
; Pseudo-frequency             	0.01
; Pseudo-frequency per oligo   	4.80769230769231e-06
; Sequence type                	DNA
; Nb of sequences              	1
; Sum of sequence lengths      	41
; discarded residues           	NA (quick mode)	 (other letters than ACGT)
; discarded occurrences        	NA (quick mode)	 (contain discarded residues)
; nb possible positions        	NA (quick mode)
; total oligo occurrences      	36
; total overlapping occurrences	2
; total non overlapping occ    	34
; alphabet size                	4
; nb possible oligomers        	2080
; oligomers tested for significance	2080
; Sequences:
;	rand	41
;
; column headers
;	1	seq            	oligomer sequence
;	2	id             	oligomer identifier
;	3	exp_freq       	expected relative frequency
;	4	occ            	observed occurrences
;	5	exp_occ        	expected occurrences
;	6	occ_P          	occurrence probability (binomial)
;	7	occ_E          	E-value for occurrences (binomial)
;	8	occ_sig        	occurrence significance (binomial)
;	9	rank           	rank
;	10	ovl_occ        	number of overlapping occurrences (discarded from the count)
;	11	forbocc        	forbidden positions (to avoid self-overlap)
#seq	id	exp_freq	occ	exp_occ	occ_P	occ_E	occ_sig	rank	ovl_occ	forbocc
aaccgt	aaccgt|acggtt	0.0000753166644	2	0.0027	3.6e-06	7.4e-03	2.13	1	0	10
cggttc	cggttc|gaaccg	0.0000840628003	2	0.003	4.4e-06	9.2e-03	2.03	2	0	10
ggttca	ggttca|tgaacc	0.0006262080504	2	0.02	0.00024	5.1e-01	0.30	3	0	10
gttcaa	gttcaa|ttgaac	0.0007418729982	2	0.03	0.00034	7.1e-01	0.15	4	0	10
; Host name	rsat
; Job started	2026-06-02.214646
; Job done	2026-06-02.214646
; Seconds	0.21
;	user	0.21
;	system	0.03
;	cuser	0.1
;	csystem	0.01
