; oligo-analysis  -v 1 -sort -i $RSAT/public_html/tmp/www-data/2026/08/21/tmp_sequence_2026-08-21.210333_k6gByF.fasta.purged -format fasta -lth occ_sig 0 -uth rank 50 -return occ,proba,rank -2str -noov -quick_if_possible -seqtype dna -bg upstream-noorf -org Drosophila_melanogaster -pseudo 0.01 -l 8 -o $RSAT/public_html/tmp/www-data/2026/08/21/oligo-analysis_2026-08-21.210333_lGYiko_8nt.tab
; Citation: van Helden et al. (1998). J Mol Biol 281(5), 827-42. 
; Program version              	1.169
; Quick counting mode          
; Detection of over-represented words (right-tail test)
; Oligomer length              	8
; Input file                   	$RSAT/public_html/tmp/www-data/2026/08/21/tmp_sequence_2026-08-21.210333_k6gByF.fasta.purged
; Input format                 	fasta
; Output file                  	$RSAT/public_html/tmp/www-data/2026/08/21/oligo-analysis_2026-08-21.210333_lGYiko_8nt.tab
; Discard overlapping matches
; Counted on both strands
; 	grouped by pairs of reverse complements
; Background model             	upstream-noorf
; Organism                     	Drosophila_melanogaster
; Background estimation method 	Frequency file
; Expected frequency file      	$RSAT/public_html/data/genomes/Drosophila_melanogaster/oligo-frequencies/8nt_upstream-noorf_Drosophila_melanogaster-noov-2str.freq
; Pseudo-frequency             	0.01
; Pseudo-frequency per oligo   	3.03988326848249e-07
; Sequence type                	DNA
; Nb of sequences              	645
; Sum of sequence lengths      	29991
; discarded residues           	NA (quick mode)	 (other letters than ACGT)
; discarded occurrences        	NA (quick mode)	 (contain discarded residues)
; nb possible positions        	NA (quick mode)
; total oligo occurrences      	25713
; total overlapping occurrences	230
; total non overlapping occ    	25483
; alphabet size                	4
; nb possible oligomers        	32896
; oligomers tested for significance	32896
;
; column headers
;	1	seq            	oligomer sequence
;	2	id             	oligomer identifier
;	3	exp_freq       	expected relative frequency
;	4	occ            	observed occurrences
;	5	exp_occ        	expected occurrences
;	6	occ_P          	occurrence probability (binomial)
;	7	occ_E          	E-value for occurrences (binomial)
;	8	occ_sig        	occurrence significance (binomial)
;	9	rank           	rank
;	10	ovl_occ        	number of overlapping occurrences (discarded from the count)
;	11	forbocc        	forbidden positions (to avoid self-overlap)
#seq	id	exp_freq	occ	exp_occ	occ_P	occ_E	occ_sig	rank	ovl_occ	forbocc
cacacaca	cacacaca|tgtgtgtg	0.0003553206980	33	9.14	8.5e-10	2.8e-05	4.55	1	20	231
aacaaaaa	aacaaaaa|tttttgtt	0.0004122277883	33	10.60	2.8e-08	9.2e-04	3.03	2	0	231
aaacaaaa	aaacaaaa|ttttgttt	0.0005055145987	37	13.00	4e-08	1.3e-03	2.88	3	2	259
acacacac	acacacac|gtgtgtgt	0.0003287893690	28	8.45	8.8e-08	2.9e-03	2.54	4	18	196
acgcgacg	acgcgacg|cgtcgcgt	0.0000166072289	6	0.43	5.8e-06	1.9e-01	0.72	5	0	42
acacagat	acacagat|atctgtgt	0.0000508748846	9	1.31	9.6e-06	3.2e-01	0.50	6	0	63
agcaacaa	agcaacaa|ttgttgct	0.0001701946718	16	4.38	1.5e-05	4.8e-01	0.32	7	0	112
aacaacaa	aacaacaa|ttgttgtt	0.0002973694410	22	7.65	1.7e-05	5.6e-01	0.25	8	0	154
acaaaaaa	acaaaaaa|ttttttgt	0.0003443806777	24	8.86	1.9e-05	6.2e-01	0.21	9	0	168
; Host name	rsat
; Job started	2026-08-21.210335
; Job done	2026-08-21.210339
; Seconds	3.34
;	user	3.34
;	system	0.08
;	cuser	0.19
;	csystem	0.03
